Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
100 changes: 100 additions & 0 deletions .github/workflows/benchmark.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,100 @@
name: Formula benchmark

on:
workflow_dispatch:
push:
branches: [codex/benchmark-rooted-accounting]
paths:
- 'benchmarks/**'
- 'testdata/llar/benchmark/**'
- '.github/workflows/benchmark.yml'

permissions:
contents: read

jobs:
benchmark:
runs-on: ubuntu-latest
timeout-minutes: 60
defaults:
run:
shell: bash
steps:
- uses: actions/checkout@v4
- name: Check native Linux and KVM
run: |
uname -a
lscpu
test -c /dev/kvm
ls -l /dev/kvm
docker info
- name: Build common payload and toolchain image
id: build
run: docker build --label "org.opencontainers.image.revision=$(git rev-parse HEAD)" -f benchmarks/Dockerfile -t sandbox-formula-bench:ci .
- name: Prepare Firecracker wrapper and identical root filesystem
id: firecracker
run: sudo bash benchmarks/setup-firecracker.sh sandbox-formula-bench:ci /tmp/firecracker-benchmark
- name: Measure three independent cold starts per backend
if: ${{ !cancelled() && steps.firecracker.outcome == 'success' }}
run: |
mkdir -p results/cold
git rev-parse HEAD > results/source-revision.txt
for trial in 1 2 3; do
case "$trial" in
1) order=firecracker,docker,sandbox ;;
2) order=docker,sandbox,firecracker ;;
3) order=sandbox,firecracker,docker ;;
esac
mkdir -p "results/cold/trial-$trial"
sudo python3 benchmarks/engine_memory.py --output "$PWD/results/cold/trial-$trial/engine-memory.jsonl" -- python3 benchmarks/run.py --image sandbox-formula-bench:ci --output "$PWD/results/cold/trial-$trial" --backends "$order" --firecracker-assets /tmp/firecracker-benchmark --concurrency 1 --builds 1 | tee "results/cold/trial-$trial/summary.jsonl"
done
- uses: actions/upload-artifact@v4
if: always()
with:
name: formula-benchmark-cold-amd64
path: results/cold/
- name: Run Firecracker Formula benchmarks
if: ${{ !cancelled() && steps.firecracker.outcome == 'success' }}
run: |
mkdir -p results/firecracker
sudo python3 benchmarks/engine_memory.py --output "$PWD/results/firecracker/engine-memory.jsonl" -- python3 benchmarks/run.py --image sandbox-formula-bench:ci --output "$PWD/results/firecracker" --backends firecracker --firecracker-assets /tmp/firecracker-benchmark --concurrency 1,2,4,8 --builds 8 | tee results/firecracker/summary.jsonl
- uses: actions/upload-artifact@v4
if: always()
with:
name: formula-benchmark-firecracker-amd64
path: results/firecracker/
- name: Run Docker Formula benchmarks
if: ${{ !cancelled() && steps.build.outcome == 'success' }}
run: |
mkdir -p results/docker
sudo python3 benchmarks/engine_memory.py --output "$PWD/results/docker/engine-memory.jsonl" -- python3 benchmarks/run.py --image sandbox-formula-bench:ci --output "$PWD/results/docker" --backends docker --concurrency 1,2,4,8 --builds 8 | tee results/docker/summary.jsonl
- uses: actions/upload-artifact@v4
if: always()
with:
name: formula-benchmark-docker-amd64
path: results/docker/
- name: Run sandbox Formula benchmarks
if: ${{ !cancelled() && steps.build.outcome == 'success' }}
run: |
mkdir -p results/sandbox
sudo python3 benchmarks/engine_memory.py --output "$PWD/results/sandbox/engine-memory.jsonl" -- python3 benchmarks/run.py --image sandbox-formula-bench:ci --output "$PWD/results/sandbox" --backends sandbox --concurrency 1,2,4,8 --builds 8 | tee results/sandbox/summary.jsonl
- name: Save exact guest kernel and VMM identities
if: always()
run: |
mkdir -p results
if test -f /tmp/firecracker-benchmark/kernel.json; then cp /tmp/firecracker-benchmark/kernel.json results/; fi
if test -f /tmp/firecracker-benchmark/checksums.txt; then cp /tmp/firecracker-benchmark/checksums.txt results/; fi
docker run --rm --entrypoint cat sandbox-formula-bench:ci /opt/benchmark/toolchain.txt > results/toolchain.txt
docker run --rm --entrypoint cat sandbox-formula-bench:ci /opt/benchmark/binaries.sha256 > results/binaries.sha256
docker run --rm --entrypoint cat sandbox-formula-bench:ci /opt/benchmark/input/manifest.json > results/input-manifest.json
for trial in 1 2 3; do
if test -f "results/cold/trial-$trial/environment.json"; then python3 benchmarks/report.py "results/cold/trial-$trial" >> "$GITHUB_STEP_SUMMARY"; fi
done
for backend in firecracker docker sandbox; do
if test -f "results/$backend/environment.json"; then python3 benchmarks/report.py "results/$backend" >> "$GITHUB_STEP_SUMMARY"; fi
done
- uses: actions/upload-artifact@v4
if: always()
with:
name: formula-benchmark-amd64
path: results/
1 change: 1 addition & 0 deletions benchmarks/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
__pycache__/
12 changes: 12 additions & 0 deletions benchmarks/Dockerfile
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
FROM golang:1.26.6-bookworm AS build
WORKDIR /src
COPY . .
RUN bash benchmarks/build.sh /out

FROM golang:1.26.6-bookworm
COPY --from=build /out /opt/benchmark
COPY benchmarks/guest-init.sh /sbin/benchmark-init
RUN chmod 755 /sbin/benchmark-init && mkdir /work && chmod 1777 /work
ENV GLIBC_TUNABLES=glibc.pthread.rseq=0 LANG=C LC_ALL=C HOME=/work TMPDIR=/tmp MAKEFLAGS=-j1
USER 1000:1000
ENTRYPOINT ["/opt/benchmark/formula-bench"]
34 changes: 34 additions & 0 deletions benchmarks/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
# LLAR Formula benchmarks

The payload is the unmodified `madler/zlib/v1.3.1/zlib_llar.gox` from xgo-dev/llarhub commit `1c2b4666ef598ed51afeb44221a73f53eea7e65c`, building zlib commit `51b7f2abdade71cd9bb0e7a373ef2610ec6f9daf`. Preparation records input hashes. Every successful sample must install a nonempty static library, header, license and pkg-config file, and return `-lz` metadata. Downloads and image construction happen before timing.

All backends call the loaded Formula's `OnBuild` directly; the `llar make` build-cache lookup is not on this path. Every job copies pristine source into its own directory on a fresh `/work` tmpfs. The controller also requires one callback start, one configure, all 19 expected source/object compilation pairs and one archive with the 15 zlib members per build. Missing or repeated work fails the run even if artifacts already exist. Normalized compiler-command hashes are retained for comparison between jobs and backends. This verifies build work; it does not claim a cold host filesystem page cache.

```sh
docker build -f benchmarks/Dockerfile -t sandbox-formula-bench:local .
python3 benchmarks/run.py --image sandbox-formula-bench:local \
--output /tmp/formula-results --backends sandbox,docker \
--concurrency 1,2,4,8 --builds 8
```

For all three backends, use a native Linux x86_64 machine with Docker, Python 3, e2fsprogs and readable/writable `/dev/kvm`:

```sh
sudo bash benchmarks/setup-firecracker.sh sandbox-formula-bench:local /tmp/firecracker-assets
sudo python3 benchmarks/run.py --image sandbox-formula-bench:local \
--output /tmp/formula-results --firecracker-assets /tmp/firecracker-assets
```

The `Formula benchmark` workflow provisions this environment on `ubuntu-latest`. Firecracker v1.17.0 and the pinned CI guest kernel 6.18.48 boot a read-only ext4 export of the exact Docker payload image, including its environment variables. It mounts private writable work/tmp directories and runs the same executable as UID/GID 1000. Kernel URL and SHA-256 are retained. No network, swap or build cache is available during execution. Every backend shares the same selected host CPU affinity, toolchain and `MAKEFLAGS=-j1`. The total memory budget defaults to 4096 MiB: sandbox shares it across its workers; Docker and Firecracker split it equally across concurrent containers. Firecracker guest RAM matches its container share; actual memory usage is measured separately.

## Measurements

- The payload comparison includes LLAR and its build tools for Docker; host LLAR, Sentry, guest and build tools for sandbox; and VMM, guest Linux, LLAR and build tools for Firecracker. The CI wrapper additionally samples `docker.service` and `containerd.service`, including their shims, and reports their sum with payload memory at each sample. These shared services are counted once per sample, not once per concurrent build. All three backends currently use an outer Docker container, so this total includes Docker services for all three. The benchmark controller and its Docker CLI clients remain excluded. Source revision, image identity, binary hashes and input hashes are retained.
- Three independent cold trials create a fresh execution environment and complete one build each. Backend order rotates across trials. These are process/VM cold starts; the host page cache is not globally flushed. `launch_to_callback_ns` uses the same callback-start boundary for all backends. The batch matrix separately measures eight builds at each concurrency with the lifecycle described below.
- `launch_to_entry_ns` is host-observed time from `docker start -a` to the entry event. Docker/Firecracker emit it at `main.main`; sandbox emits it at the redirected guest entry. All three include their outer Docker launch. For sandbox this also includes host Formula preparation and export. Marker delivery overhead is included; these are not bare-kernel boot times.
- Serial sandbox samples additionally report `entry_ns`: host-observed Run-start to guest-entry markers, after Go initialization but before `state.Load`, and `callback_ns`: Run-start to the first closure instruction after restoration. The benchmark-only Go overlay adds a stdout marker at guest entry. Library sources and production behavior are unchanged; no syscall inspector is enabled. All backends send markers through stdout, so delivery overhead is included. Parallel sandbox runs do not guess which Run an entry belongs to. Direct workers also emit a callback-start event so application readiness can be distinguished from reaching main.
- `duration_ns` times `OnBuild` plus validation; sandbox includes state export, guest execution and writeback. The batch throughput includes Formula preparation, container startup, execution, exit and collection, but ends before container deletion. Raw events distinguish these phases.
- The controller samples raw Docker API `memory_stats.usage` with a 100 ms wait between rounds and sums active containers. The optional `engine_memory.py` wrapper reads disjoint cgroup-v2 `memory.current` counters for payloads and Docker services on the native systemd runner, preserving baseline, per-sample breakdowns and runtime-process membership. Totals come from paired samples, not the sum of separate peaks. Both metrics include page cache and remain separate from cache-subtracted Docker CLI displays and Go heap counters. Short peaks can be missed and reads across cgroups are not atomic. State Save/Load and exit GC remain inside the measurement window.
- Every concurrency level completes the same requested number of builds. Each sandbox worker owns one interpreter, reused sequentially with fresh build contexts and directories; workers share one Kernel. Docker and Firecracker create one execution environment per build. Interpreter construction finishes before concurrent sandbox transfers, as required by the current type-cache contract. Failures remain failures and are excluded from successful latency percentiles.

Each run writes environment metadata, container logs, per-task JSON, raw memory samples and batch summaries. Do not compare local Docker Desktop ARM64 results with native amd64 runner results as if they came from the same machine.
28 changes: 28 additions & 0 deletions benchmarks/build.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
#!/usr/bin/env bash
set -euo pipefail
repo=$(cd "$(dirname "$0")/.." && pwd)
mkdir -p "$1"
output=$(cd "$1" && pwd)
python3 "$repo/benchmarks/prepare.py" "$output/input"
bash "$repo/sentry/build-linux.sh" "$output"
# Instrument only the benchmark build. Production guest entry stays unchanged.
python3 - "$repo" "$output" <<'PY'
import json, pathlib, sys
repo, output = map(pathlib.Path, sys.argv[1:])
source = repo / "guest_linux.go"
text = source.read_text()
entry = "func guestEntry() {\n"
assert text.count(entry) == 1
replacement = output / "guest_linux.go"
replacement.write_text(text.replace(entry, entry + '\tif _, err := os.Stdout.WriteString("BENCH:{\\"event\\":\\"guest_entry\\"}\\n"); err != nil { panic(err) }\n'))
(output / "overlay.json").write_text(json.dumps({"Replace": {str(source): str(replacement)}}))
PY
go -C "$repo/testdata/llar" build -mod=readonly -overlay="$output/overlay.json" \
-ldflags='-checklinkname=0 -s=false -w=false -extldflags=-Wl,-z,separate-code' -o "$output/formula-bench" ./benchmark
gcc -O2 -o "$output/reboot" "$repo/benchmarks/reboot.c"
chmod 755 "$output" "$output/formula-bench"
go version > "$output/toolchain.txt"
gcc --version >> "$output/toolchain.txt"
make --version >> "$output/toolchain.txt"
pkg-config --version >> "$output/toolchain.txt"
sha256sum "$output/formula-bench" "$output/sentrylib.so" > "$output/binaries.sha256"
132 changes: 132 additions & 0 deletions benchmarks/engine_memory.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,132 @@
#!/usr/bin/env python3
"""Include the shared Docker services without changing the measured payload."""
import argparse
import json
import pathlib
import subprocess
import threading
import time


def main():
parser = argparse.ArgumentParser()
parser.add_argument("--output", required=True)
parser.add_argument("command", nargs=argparse.REMAINDER)
args = parser.parse_args()
command = args.command
if command and command[0] == "--":
command = command[1:]
if not command:
parser.error("a benchmark command is required")

info = json.loads(subprocess.check_output(["docker", "info", "--format", "{{json .}}"], text=True))
if info["CgroupDriver"] != "systemd" or info["CgroupVersion"] != "2" or info["Containers"] != 0:
raise RuntimeError("engine accounting requires an idle native Linux systemd/cgroup-v2 Docker daemon")
root = pathlib.Path("/sys/fs/cgroup")
services = {}
for service in ("docker.service", "containerd.service"):
group = subprocess.check_output(["systemctl", "show", "--property=ControlGroup", "--value", service], text=True).strip()
if not group or group == "/":
raise RuntimeError(f"missing service cgroup: {service}")
services[service] = root / group.lstrip("/")
paths = list(services.values())
if paths[0] == paths[1] or paths[0] in paths[1].parents or paths[1] in paths[0].parents:
raise RuntimeError("Docker service cgroups overlap")

output = pathlib.Path(args.output)
output.parent.mkdir(parents=True, exist_ok=True)
samples, errors = [], []
finished = threading.Event()

def memory(path):
return {"bytes": int((path / "memory.current").read_text()),
"stats": dict((key, int(value)) for key, value in
(line.split() for line in (path / "memory.stat").read_text().splitlines()))}

def sample():
begin = time.perf_counter_ns()
engine = {name: memory(path) for name, path in services.items()}
containers = {}
for path in (root / "system.slice").glob("docker-*.scope"):
cid = path.name.removeprefix("docker-").removesuffix(".scope")
if len(cid) != 64:
raise RuntimeError(f"unexpected container cgroup: {path}")
if any(service == path or service in path.parents or path in service.parents for service in paths):
raise RuntimeError(f"container and service cgroups overlap: {path}")
try:
containers[cid] = memory(path)
except FileNotFoundError:
# A completed container can disappear between the directory
# listing and the read. Service cgroups must remain available.
if path.exists():
raise
runtime_processes = []
for proc in pathlib.Path("/proc").iterdir():
if not proc.name.isdecimal():
continue
try:
name = (proc / "comm").read_text().strip()
if name not in ("dockerd", "containerd") and not name.startswith("containerd-shim"):
continue
group = (proc / "cgroup").read_text().strip().removeprefix("0::")
except FileNotFoundError:
continue
path = root / group.lstrip("/")
if not any(service == path or service in path.parents for service in paths):
raise RuntimeError(f"unaccounted Docker runtime process: {proc.name} {name} {group}")
runtime_processes.append({"pid": int(proc.name), "name": name, "cgroup": group})
entry = {"begin_ns": begin, "end_ns": time.perf_counter_ns(),
"services": engine, "containers": containers, "runtime_processes": runtime_processes,
"engine_bytes": sum(value["bytes"] for value in engine.values()),
"payload_bytes": sum(value["bytes"] for value in containers.values())}
entry["total_bytes"] = entry["engine_bytes"] + entry["payload_bytes"]
samples.append(entry)
log.write(json.dumps(entry) + "\n")
log.flush()

def observe():
try:
while not finished.wait(0.1):
sample()
except Exception as error:
errors.append(str(error))

with output.open("w") as log:
sample()
observer = threading.Thread(target=observe)
observer.start()
try:
result = subprocess.run(command)
finally:
finished.set()
observer.join()
sample()

for path in sorted(output.parent.glob("*-c*.json")):
data = json.loads(path.read_text())
if "records" not in data:
continue
start = min(record["start_ns"] - record["create_ns"] for record in data["records"])
end = max(record["start_ns"] + record["wall_ns"] for record in data["records"])
ids = {record["container"] for record in data["records"]}
window = [entry for entry in samples if entry["begin_ns"] <= end and entry["end_ns"] >= start]
if not window or not any(ids.intersection(entry["containers"]) for entry in window):
errors.append(f"no payload cgroup samples for {path.name}")
data["engine_accounting"] = {
"services": {name: str(path) for name, path in services.items()},
"baseline_bytes": samples[0]["engine_bytes"],
"engine_peak_bytes": max((entry["engine_bytes"] for entry in window), default=0),
"payload_peak_bytes": max((sum(value["bytes"] for cid, value in entry["containers"].items() if cid in ids) for entry in window), default=0),
"total_peak_bytes": max((entry["engine_bytes"] + sum(value["bytes"] for cid, value in entry["containers"].items() if cid in ids) for entry in window), default=0),
"samples": len(window), "errors": list(errors),
"metric": "Sum of disjoint cgroup-v2 memory.current reads in each sample; includes cache and shared Docker services once",
"excluded": "benchmark controller and its Docker CLI clients; host kernel outside these cgroups",
}
path.write_text(json.dumps(data, indent=2) + "\n")
if errors:
raise SystemExit("engine memory sampling failed: " + "; ".join(errors))
raise SystemExit(result.returncode)


if __name__ == "__main__":
main()
Loading
Loading