diff --git a/.github/workflows/test-ls-on-firecracker.yml b/.github/workflows/test-ls-on-firecracker.yml new file mode 100644 index 0000000..c9b4939 --- /dev/null +++ b/.github/workflows/test-ls-on-firecracker.yml @@ -0,0 +1,56 @@ +name: LocalStack on Firecracker + +on: + pull_request: + branches: [main] + paths: + - 'ls-on-firecracker/**' + - '.github/workflows/test-ls-on-firecracker.yml' + push: + branches: [main] + paths: + - 'ls-on-firecracker/**' + - '.github/workflows/test-ls-on-firecracker.yml' + workflow_dispatch: + +env: + LOCALSTACK_AUTH_TOKEN: ${{ secrets.TEST_LOCALSTACK_AUTH_TOKEN }} + +jobs: + test-ls-on-firecracker: + name: Firecracker microVM + LocalStack + S3/Lambda smoke test + runs-on: ubuntu-latest + timeout-minutes: 30 + defaults: + run: + working-directory: ls-on-firecracker + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Enable KVM access + run: | + echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' \ + | sudo tee /etc/udev/rules.d/99-kvm4all.rules + sudo udevadm control --reload-rules + sudo udevadm trigger --name-match=kvm + ls -l /dev/kvm + + - name: Boot the microVM + run: make up + + - name: Run S3 + Lambda smoke test + run: make test + + - name: Dump console/boot log on failure + if: failure() + run: cat work/firecracker.log 2>/dev/null || true + + - name: Dump service diagnostics on failure + if: failure() + run: make diagnose || true + + - name: Tear down + if: always() + run: make down diff --git a/ls-on-firecracker/.gitignore b/ls-on-firecracker/.gitignore new file mode 100644 index 0000000..9d931c4 --- /dev/null +++ b/ls-on-firecracker/.gitignore @@ -0,0 +1 @@ +work/ diff --git a/ls-on-firecracker/Makefile b/ls-on-firecracker/Makefile new file mode 100644 index 0000000..6e8b80d --- /dev/null +++ b/ls-on-firecracker/Makefile @@ -0,0 +1,60 @@ +FC_VERSION := v1.10.1 +CI_TRACK := v1.10 +UNAME_M := $(shell uname -m) +# Firecracker's release/CI artifact names use "aarch64"/"x86_64", not the +# "arm64"/"amd64" spellings uname reports on macOS or Debian-flavored Linux. +ARCH := $(if $(filter arm64,$(UNAME_M)),aarch64,$(if $(filter amd64,$(UNAME_M)),x86_64,$(UNAME_M))) +WORK_DIR := work +TAP_DEV := fc-ls-tap0 +TAP_IP := 172.16.0.1 +VM_IP := 172.16.0.2 +VM_MASK := 255.255.255.0 +VCPUS := 2 +MEM_MB := 4096 +BUCKET := firecracker-demo +LAMBDA_FN := firecracker-demo-fn +LAMBDA_BUCKET := firecracker-demo-lambda-bucket + +export WORK_DIR TAP_DEV TAP_IP VM_IP VM_MASK VCPUS MEM_MB BUCKET LAMBDA_FN LAMBDA_BUCKET FC_VERSION CI_TRACK ARCH LOCALSTACK_AUTH_TOKEN + +SSH := ssh -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o LogLevel=ERROR \ + -o BatchMode=yes -o ConnectTimeout=10 \ + -i $(WORK_DIR)/images/id_rsa root@$(VM_IP) + +.PHONY: help download rootfs up test logs down clean ssh diagnose + +help: ## Show available targets + @grep -E '^[a-zA-Z_-]+:.*##' $(MAKEFILE_LIST) | \ + awk 'BEGIN{FS=":.*##"}{printf " %-10s %s\n", $$1, $$2}' + +download: ## Fetch the firecracker binary, guest kernel and base rootfs + @bash scripts/download-assets.sh + +rootfs: download ## Build a LocalStack-flavored guest rootfs image + @bash scripts/build-rootfs.sh + +up: rootfs ## Create the tap device and boot the LocalStack microVM + @bash scripts/run-vm.sh + +test: ## Run S3 + Lambda smoke tests against the running LocalStack instance + @bash scripts/smoke-test.sh + +logs: ## Tail the microVM's console/boot log + @tail -n 100 -f $(WORK_DIR)/firecracker.log + +ssh: ## SSH into the running microVM as root + @$(SSH) + +diagnose: ## Dump docker/localstack service status + logs from inside the microVM + @$(SSH) 'systemctl --no-pager status docker.service localstack.service; \ + echo ---localstack.service exit info---; \ + systemctl show localstack.service -p Result -p ExecMainStatus -p ExecMainCode; \ + echo ---docker.journal---; journalctl -u docker --no-pager -n 100; \ + echo "---localstack.journal (single attempt, Restart=no)---"; \ + journalctl -u localstack --no-pager' + +down: ## Stop the microVM and remove the tap device + @bash scripts/teardown.sh + +clean: down ## Remove all downloaded/built artifacts + rm -rf $(WORK_DIR) diff --git a/ls-on-firecracker/README.md b/ls-on-firecracker/README.md new file mode 100644 index 0000000..80db906 --- /dev/null +++ b/ls-on-firecracker/README.md @@ -0,0 +1,48 @@ +# LocalStack on Firecracker + +A demo that boots [LocalStack](https://localstack.cloud) inside a +[Firecracker](https://firecracker-microvm.github.io/) microVM — the same +technology AWS Lambda itself runs on — and exercises it over the network with +an S3 bucket and a Lambda function that itself talks back to S3. + +## Quick start + +``` +make up # download -> build a LocalStack-flavored rootfs -> boot the microVM +make test # S3 round-trip + Lambda deploy/invoke through it +make down # stop the VM +``` + +Run `make help` for the full target list, including `make ssh` and +`make diagnose` for poking around inside the running microVM. + +## Prerequisites + +- A **Linux host with KVM** (`/dev/kvm`) — doesn't run on macOS or most cloud + VMs without nested virtualization. Bare metal, an EC2 `.metal` instance, or + a GitHub Actions `ubuntu-latest` runner all work; see + `.github/workflows/test-ls-on-firecracker.yml` for a working CI setup. +- `curl`, `iproute2`, `iptables`, `e2fsprogs`, `zip`, `jq`, `ssh`, and the + AWS CLI on the host, plus `sudo` access. +- A `LOCALSTACK_AUTH_TOKEN` environment variable set to a LocalStack **CI + Auth Token** — get one from + [your LocalStack workspace](https://app.localstack.cloud/workspace/auth-tokens). + +## Layout + +``` +Makefile self-describing entry point (make help) +scripts/ + download-assets.sh fetch firecracker + lstk + kernel + base rootfs + build-rootfs.sh install Docker + lstk into a working rootfs image + run-vm.sh set up networking (incl. NAT) and boot the microVM + smoke-test.sh S3 round-trip + Lambda deploy/invoke against it + teardown.sh stop the VM and remove the tap device / NAT rules +fixtures/ + handler.py the demo Lambda function +docs/ + NOTES.md how it works under the hood, and known caveats +work/ downloaded/generated artifacts (git-ignored) +``` + +See [docs/NOTES.md](docs/NOTES.md) for the details. diff --git a/ls-on-firecracker/docs/NOTES.md b/ls-on-firecracker/docs/NOTES.md new file mode 100644 index 0000000..66b3510 --- /dev/null +++ b/ls-on-firecracker/docs/NOTES.md @@ -0,0 +1,66 @@ +# How it works + +1. **`make download`** grabs the `firecracker` binary, LocalStack's own + [`lstk`](https://docs.localstack.cloud/aws/developer-tools/running-localstack/lstk/) + CLI, a guest kernel and a base Ubuntu rootfs from Firecracker's public CI + artifacts. +2. **`make rootfs`** clones the base rootfs, grows it, and chroots in to + install Docker and drop in the `lstk` binary, then registers `systemd` + units so Docker and then `lstk start` come up on boot. Building happens on + the host via a loop-mounted image, so the customization step itself + doesn't need the guest to be running. +3. **`make up`** creates a tap network device on the host, NATs the guest out + through the host's default interface (`lstk` needs to pull the LocalStack + image, and LocalStack itself needs to pull the Lambda runtime image at + invoke time), and boots the image with Firecracker. It polls + `http://:4566/_localstack/health` until LocalStack is ready. +4. **`make test`** creates an S3 bucket and round-trips an object, then + deploys a small Python Lambda function and invokes it. The function + itself creates a *second* bucket and lists all buckets via `boto3` + (LocalStack injects `AWS_ENDPOINT_URL` into the Lambda execution + environment automatically, so no endpoint code is needed) — proving the + Lambda's own AWS calls land on the same LocalStack backend as the CLI + calls above: it sees the first bucket, and the one it creates is visible + back on the CLI afterward. `lstk` runs LocalStack as a container against + the guest's own Docker daemon, which is also what LocalStack itself uses + to spawn the Lambda executor container — the same two-layer shape real + Lambda uses (a container runtime inside a Firecracker microVM). +5. **`make down`** / **`make clean`** tear the VM, NAT rules, and tap device + down again. + +`make ssh` drops you into a root shell on the running microVM (the base +image ships a pre-authorized SSH key, fetched by `download-assets.sh` +alongside the kernel/rootfs). `make diagnose` dumps `systemctl status` and +`journalctl` for the `docker`/`localstack` services — the same thing the CI +workflow does automatically on failure. + +## What's actually running + +Docker runs *inside* the guest OS (installed at rootfs-build time), and +`lstk` uses it to pull and run the LocalStack container, which in turn uses +the same Docker daemon as its normal Docker-based Lambda executor. The guest +needs internet access both to pull the LocalStack image and, at Lambda +invoke time, the runtime image — which is why `run-vm.sh` sets up NAT +through the host rather than an isolated host-only network. + +## Caveats + +The base rootfs comes from Firecracker's own **CI test artifacts** — what +their integration tests boot, not a general-purpose image. It's stripped +down accordingly (no `/var/cache/apt`, `/var/log`, or populated dpkg +database out of the box; `build-rootfs.sh` reconstructs what apt/Docker +need), and its kernel is minimal too (no loadable modules, no `nf_tables`, +no `CONFIG_IP_NF_RAW`) — a few of the fixes in `build-rootfs.sh` exist +specifically to work around that (switching Docker to the legacy iptables +backend, and opting out of a Docker 28+ hardening rule that needs a table +this kernel doesn't have via `DOCKER_INSECURE_NO_IPTABLES_RAW=1`, which is +fine for a single-tenant, ephemeral microVM but not something to carry into +a shared or long-lived host). + +A more idiomatic base for "run a Docker image as a Firecracker rootfs" would +be `docker export`-ing a real image (e.g. `ubuntu:22.04`) onto a formatted +ext4 device rather than patching Firecracker's CI artifact. For running +actual container workloads inside Firecracker in production, see +[firecracker-containerd](https://github.com/firecracker-microvm/firecracker-containerd) +(what AWS Lambda/Fargate use) instead of a full Docker-in-VM setup like this +one. diff --git a/ls-on-firecracker/fixtures/handler.py b/ls-on-firecracker/fixtures/handler.py new file mode 100644 index 0000000..183dfe8 --- /dev/null +++ b/ls-on-firecracker/fixtures/handler.py @@ -0,0 +1,17 @@ +import boto3 + +# LocalStack injects AWS_ENDPOINT_URL into the Lambda execution environment +# automatically, and boto3 has respected that variable since 1.28.0 -- no +# endpoint_url override needed here, on LocalStack or on real AWS. +s3 = boto3.client("s3") + + +def handler(event, context): + name = event.get("name", "world") + bucket = event.get("bucket") + + if bucket: + s3.create_bucket(Bucket=bucket) + + buckets = [b["Name"] for b in s3.list_buckets()["Buckets"]] + return {"message": f"hello {name}", "buckets": buckets} diff --git a/ls-on-firecracker/scripts/build-rootfs.sh b/ls-on-firecracker/scripts/build-rootfs.sh new file mode 100755 index 0000000..f9ebe3b --- /dev/null +++ b/ls-on-firecracker/scripts/build-rootfs.sh @@ -0,0 +1,177 @@ +#!/usr/bin/env bash +# Turns the pristine CI rootfs into a "LocalStack appliance": grows the ext4 +# image, chroots into it to install Docker and drop in the lstk binary, and +# registers a systemd unit that runs `lstk start` (which pulls and runs the +# LocalStack container against the guest's own Docker daemon) on boot. +# +# The chroot shares the host's network namespace (it's just a mounted +# directory, not a container), so apt-get works normally here as long as the +# host has internet access. The guest VM itself needs its own internet +# access too, at boot time this time -- lstk has to pull the LocalStack +# image and Lambda invocations pull runtime images -- which is what the NAT +# setup in run-vm.sh is for. +set -euo pipefail + +: "${WORK_DIR:?run via 'make', not directly}" + +IMG_DIR="$WORK_DIR/images" +BASE_IMG="$IMG_DIR/base.ext4" +OUT_IMG="$IMG_DIR/localstack.ext4" +MNT="$WORK_DIR/rootfs-mnt" + +if [[ -f "$OUT_IMG" ]]; then + echo "[rootfs] $OUT_IMG already built, skipping (delete it, or 'make clean', to rebuild)" + exit 0 +fi + +echo "[rootfs] cloning base image and growing it to make room for LocalStack + Docker" +cp "$BASE_IMG" "$OUT_IMG" +truncate -s 8G "$OUT_IMG" +e2fsck -fy "$OUT_IMG" || true +resize2fs "$OUT_IMG" + +mkdir -p "$MNT" +LOOP_DEV=$(sudo losetup --find --show "$OUT_IMG") +cleanup() { + sudo umount -R "$MNT" 2>/dev/null || true + sudo losetup -d "$LOOP_DEV" 2>/dev/null || true +} +trap cleanup EXIT + +sudo mount "$LOOP_DEV" "$MNT" +sudo cp /etc/resolv.conf "$MNT/etc/resolv.conf" +sudo mount --bind /dev "$MNT/dev" +sudo mount --bind /proc "$MNT/proc" +sudo mount --bind /sys "$MNT/sys" +# The base image's /tmp and /run are normally populated by systemd-tmpfiles +# at boot; without that, apt/dpkg (which write scratch files there) fail +# with confusing "No such file or directory" errors inside the chroot. +sudo mkdir -p "$MNT/tmp" "$MNT/run" +sudo mount -t tmpfs tmpfs "$MNT/tmp" +sudo mount -t tmpfs tmpfs "$MNT/run" +sudo chmod 1777 "$MNT/tmp" + +# This CI-provided base image ships apt/dpkg binaries, but it was stripped +# for Firecracker's own network-test use, not general package installs: +# /var/cache/apt, /var/lib/apt and /var/log don't exist at all, and +# /var/lib/dpkg is an empty directory with no status file. Recreate the +# standard skeleton apt/dpkg expect -- the same bootstrap debootstrap itself +# does for a fresh root -- before touching either. +sudo mkdir -p \ + "$MNT/var/cache/apt/archives/partial" \ + "$MNT/var/lib/apt/lists/partial" \ + "$MNT/var/log/apt" \ + "$MNT/var/lib/dpkg/info" \ + "$MNT/var/lib/dpkg/updates" \ + "$MNT/var/lib/dpkg/triggers" \ + "$MNT/var/backups" +sudo touch "$MNT/var/lib/dpkg/status" "$MNT/var/lib/dpkg/available" + +echo "[rootfs] installing Docker + LocalStack inside the guest image (this can take a few minutes)" +sudo chroot "$MNT" /bin/bash -c ' + set -euo pipefail + export DEBIAN_FRONTEND=noninteractive + apt-get update -qq + # --force-confdef/--force-confold: some packages (libpam-modules and + # friends) think their conffiles were locally modified in this base image + # and prompt for a merge decision on stdin, which is not a TTY here. + apt-get install -y -qq \ + -o Dpkg::Options::=--force-confdef \ + -o Dpkg::Options::=--force-confold \ + docker.io >/dev/null + # The Firecracker CI kernel does not compile in nf_tables (only the + # legacy x_tables framework), but Ubuntu 22.04 iptables defaults to the + # nftables backend -- dockerd then fails at startup with "Failed to + # initialize nft: Protocol not supported". Switch to the legacy backend, + # which the kernel does support. + update-alternatives --set iptables /usr/sbin/iptables-legacy + update-alternatives --set ip6tables /usr/sbin/ip6tables-legacy + systemctl enable docker.service +' + +# Docker 28+ adds a DROP rule in the iptables "raw" table as a hardening +# measure (prevents reaching a container directly, bypassing its published +# port restriction). That table needs CONFIG_IP_NF_RAW, which this kernel +# does not have -- and can't load at runtime either, since module loading +# is compiled out entirely. DOCKER_INSECURE_NO_IPTABLES_RAW=1 (Docker +# 28.0.2+) is the documented opt-out for exactly this case. The tradeoff +# (a container published to 127.0.0.1 becomes reachable from other hosts +# on the same network) is acceptable here: this microVM is single-tenant, +# ephemeral, and only reachable via the host-only tap network run-vm.sh +# sets up, not by "other hosts on the local network". +sudo mkdir -p "$MNT/etc/systemd/system/docker.service.d" +sudo tee "$MNT/etc/systemd/system/docker.service.d/no-iptables-raw.conf" >/dev/null <<'UNIT' +[Service] +Environment=DOCKER_INSECURE_NO_IPTABLES_RAW=1 +UNIT + +sudo cp "$WORK_DIR/bin/lstk" "$MNT/usr/local/bin/lstk" +sudo chmod +x "$MNT/usr/local/bin/lstk" + +# lstk publishes the emulator's port bound to the *guest's* loopback by +# default (127.0.0.1) -- fine when lstk and its caller are on the same host, +# but we reach the emulator from outside the guest, over the tap network. +# The bind host comes from the first entry of GATEWAY_LISTEN, which lstk only +# reads from a config.toml [env.*] profile (a plain LOCALSTACK_GATEWAY_LISTEN +# process env var reaches the container too late, after the host-side Docker +# port binding is already decided). $HOME is /root (set above), so this is +# the second entry in lstk's config search order. +sudo mkdir -p "$MNT/root/.config/lstk" +sudo tee "$MNT/root/.config/lstk/config.toml" >/dev/null <<'TOML' +[[containers]] +type = "aws" +port = "4566" +env = ["net"] + +[env.net] +GATEWAY_LISTEN = "0.0.0.0:4566,0.0.0.0:443" +TOML + +# lstk pulls the LocalStack image and starts it as a container against the +# guest's own Docker daemon -- the container is what actually runs LocalStack +# and spawns Lambda executor containers, exactly like real Lambda uses a +# container runtime inside its Firecracker microVM. `lstk start` blocks +# until the emulator is ready and then exits, so the unit that "is" this +# service is really the container, not this process -- hence oneshot + +# RemainAfterExit rather than a long-running ExecStart. +sudo tee "$MNT/etc/systemd/system/localstack.service" >/dev/null <<'UNIT' +[Unit] +Description=LocalStack (via lstk) +After=docker.service network-online.target +Wants=network-online.target +Requires=docker.service + +[Service] +Type=oneshot +RemainAfterExit=yes +# systemd services don't get $HOME the way login shells do, but lstk needs +# it to resolve its config/cache directory (~/.cache/lstk/...); this unit +# runs as root (no User= override), so point it at root's home. +Environment=HOME=/root +ExecStart=/usr/local/bin/lstk start --non-interactive --timeout 120s +TimeoutStartSec=200 +# No auto-restart: a single clear failure (visible via `systemctl is-failed` +# and its journal) is far more useful for a demo/CI than systemd silently +# retrying a broken command every few seconds for the entire boot budget, +# burying the real error under repeated "Failed to start" lines. Re-run +# `make up` (which boots a fresh VM) to retry. +Restart=no + +[Install] +WantedBy=multi-user.target +UNIT + +sudo chroot "$MNT" systemctl enable localstack.service + +# The chroot borrowed the host's /etc/resolv.conf (often a systemd-resolved +# stub at 127.0.0.53) to resolve apt mirrors during the build above. That +# address is meaningless once the image boots as its own VM, so pin a public +# resolver for runtime -- the guest needs it to pull the LocalStack image +# (lstk) and the Lambda runtime image (LocalStack itself) over the NAT'd +# link run-vm.sh sets up. +sudo tee "$MNT/etc/resolv.conf" >/dev/null <<'EOF' +nameserver 8.8.8.8 +nameserver 1.1.1.1 +EOF + +echo "[rootfs] built -> $OUT_IMG" diff --git a/ls-on-firecracker/scripts/download-assets.sh b/ls-on-firecracker/scripts/download-assets.sh new file mode 100755 index 0000000..f743067 --- /dev/null +++ b/ls-on-firecracker/scripts/download-assets.sh @@ -0,0 +1,101 @@ +#!/usr/bin/env bash +# Downloads the three things Firecracker needs to boot a microVM: +# 1. the firecracker binary itself +# 2. an uncompressed guest kernel (vmlinux) +# 3. a base guest rootfs (ext4 image) +# +# Kernel/rootfs are pulled from Firecracker's public CI bucket, which is the +# same source used in the project's own getting-started guide. We resolve +# "latest for this CI track" dynamically so the URLs don't go stale. +set -euo pipefail + +: "${WORK_DIR:?run via 'make', not directly}" +: "${FC_VERSION:?}" +: "${CI_TRACK:?}" +: "${ARCH:?}" + +if [[ "$(uname -s)" != "Linux" ]]; then + echo "[download] error: Firecracker requires Linux + KVM (/dev/kvm)." >&2 + echo " This host is $(uname -s), so it cannot run this demo." >&2 + echo " Try a Linux box with KVM, an EC2 .metal instance, or a" >&2 + echo " GitHub Actions ubuntu-latest runner instead." >&2 + exit 1 +fi + +if [[ ! -e /dev/kvm ]]; then + echo "[download] error: /dev/kvm not found. Firecracker needs KVM, which" >&2 + echo " usually means bare metal or a host with nested" >&2 + echo " virtualization enabled (not a typical cloud VM)." >&2 + exit 1 +fi + +BIN_DIR="$WORK_DIR/bin" +IMG_DIR="$WORK_DIR/images" +mkdir -p "$BIN_DIR" "$IMG_DIR" + +if [[ -x "$BIN_DIR/firecracker" ]]; then + echo "[download] firecracker binary already present, skipping" +else + echo "[download] fetching firecracker $FC_VERSION for $ARCH" + tmp=$(mktemp -d) + curl -fsSL "https://github.com/firecracker-microvm/firecracker/releases/download/${FC_VERSION}/firecracker-${FC_VERSION}-${ARCH}.tgz" \ + | tar -xz -C "$tmp" + find "$tmp" -type f -name "firecracker-${FC_VERSION}-${ARCH}" -exec cp {} "$BIN_DIR/firecracker" \; + chmod +x "$BIN_DIR/firecracker" + rm -rf "$tmp" +fi + +# lstk is LocalStack's own CLI: it pulls the LocalStack image, starts it as a +# container against the guest's Docker daemon, and waits for it to be ready. +# It ends up baked into the guest rootfs (see build-rootfs.sh), not run on +# the host, so we resolve its release for the *guest's* architecture. +GOARCH="$(if [[ "$ARCH" == "aarch64" ]]; then echo arm64; else echo amd64; fi)" +if [[ -x "$BIN_DIR/lstk" ]]; then + echo "[download] lstk binary already present, skipping" +else + echo "[download] fetching latest lstk for linux/$GOARCH" + lstk_tag=$(curl -fsSLI -o /dev/null -w '%{url_effective}' "https://github.com/localstack/lstk/releases/latest" | sed 's#.*/##') + [[ -n "$lstk_tag" ]] || { echo "could not resolve the latest lstk release" >&2; exit 1; } + lstk_ver="${lstk_tag#v}" + tmp=$(mktemp -d) + curl -fsSL "https://github.com/localstack/lstk/releases/download/${lstk_tag}/lstk_${lstk_ver}_linux_${GOARCH}.tar.gz" \ + | tar -xz -C "$tmp" lstk + mv "$tmp/lstk" "$BIN_DIR/lstk" + chmod +x "$BIN_DIR/lstk" + rm -rf "$tmp" +fi + +if [[ -f "$IMG_DIR/vmlinux.bin" ]]; then + echo "[download] kernel already present, skipping" +else + echo "[download] resolving latest CI kernel for track $CI_TRACK/$ARCH" + kernel_key=$(curl -fsSL "http://spec.ccfc.min.s3.amazonaws.com/?prefix=firecracker-ci/${CI_TRACK}/${ARCH}/vmlinux-&list-type=2" \ + | grep -oP '(?<=)[^<]+' \ + | grep -E "^firecracker-ci/${CI_TRACK}/${ARCH}/vmlinux-[0-9]+\.[0-9]+\.[0-9]+$" \ + | sort -V | tail -1) + [[ -n "$kernel_key" ]] || { echo "could not resolve a kernel for CI track $CI_TRACK/$ARCH" >&2; exit 1; } + curl -fsSL "https://s3.amazonaws.com/spec.ccfc.min/${kernel_key}" -o "$IMG_DIR/vmlinux.bin" +fi + +if [[ -f "$IMG_DIR/base.ext4" ]]; then + echo "[download] base rootfs already present, skipping" +else + echo "[download] resolving latest CI rootfs for track $CI_TRACK/$ARCH" + rootfs_key=$(curl -fsSL "http://spec.ccfc.min.s3.amazonaws.com/?prefix=firecracker-ci/${CI_TRACK}/${ARCH}/ubuntu-&list-type=2" \ + | grep -oP '(?<=)[^<]+' \ + | grep -E "^firecracker-ci/${CI_TRACK}/${ARCH}/ubuntu-[0-9]+\.[0-9]+\.ext4$" \ + | sort -V | tail -1) + [[ -n "$rootfs_key" ]] || { echo "could not resolve a base rootfs for CI track $CI_TRACK/$ARCH" >&2; exit 1; } + curl -fsSL "https://s3.amazonaws.com/spec.ccfc.min/${rootfs_key}" -o "$IMG_DIR/base.ext4" + + # This base image ships sshd running with a pre-authorized root key; the + # matching private key is published as a sibling of the .ext4 file (e.g. + # ubuntu-22.04.id_rsa next to ubuntu-22.04.ext4), not an appended suffix. + # Handy for `make ssh` and pulling diagnostics when a boot fails. + rsa_key="${rootfs_key%.ext4}.id_rsa" + curl -fsSL "https://s3.amazonaws.com/spec.ccfc.min/${rsa_key}" -o "$IMG_DIR/id_rsa" \ + && chmod 600 "$IMG_DIR/id_rsa" \ + || echo "[download] warning: no matching SSH key found for this rootfs, 'make ssh' won't work" >&2 +fi + +echo "[download] done -> $BIN_DIR/firecracker, $BIN_DIR/lstk, $IMG_DIR/vmlinux.bin, $IMG_DIR/base.ext4" diff --git a/ls-on-firecracker/scripts/run-vm.sh b/ls-on-firecracker/scripts/run-vm.sh new file mode 100755 index 0000000..8a8e455 --- /dev/null +++ b/ls-on-firecracker/scripts/run-vm.sh @@ -0,0 +1,142 @@ +#!/usr/bin/env bash +# Sets up a tap device on the host, writes a Firecracker VM config, and boots +# the microVM in the background. Waits until LocalStack answers its health +# endpoint over the tap network before returning. +set -euo pipefail + +: "${WORK_DIR:?run via 'make', not directly}" +: "${TAP_DEV:?}" "${TAP_IP:?}" "${VM_IP:?}" "${VM_MASK:?}" "${VCPUS:?}" "${MEM_MB:?}" + +FC_BIN="$WORK_DIR/bin/firecracker" +KERNEL="$WORK_DIR/images/vmlinux.bin" +ROOTFS="$WORK_DIR/images/localstack.ext4" +SOCKET="$WORK_DIR/firecracker.sock" +CONFIG="$WORK_DIR/vm-config.json" +LOG="$WORK_DIR/firecracker.log" +PIDFILE="$WORK_DIR/firecracker.pid" + +if [[ -f "$PIDFILE" ]] && kill -0 "$(cat "$PIDFILE")" 2>/dev/null; then + echo "[run] a microVM is already running (pid $(cat "$PIDFILE")); run 'make down' first" + exit 0 +fi +rm -f "$SOCKET" + +echo "[run] configuring tap device $TAP_DEV ($TAP_IP <-> $VM_IP)" +if ! ip link show "$TAP_DEV" &>/dev/null; then + sudo ip tuntap add dev "$TAP_DEV" mode tap +fi +sudo ip addr flush dev "$TAP_DEV" +sudo ip addr add "${TAP_IP}/24" dev "$TAP_DEV" +sudo ip link set dev "$TAP_DEV" up +sudo sysctl -qw "net.ipv4.conf.${TAP_DEV}.proxy_arp=1" +sudo sysctl -qw "net.ipv6.conf.${TAP_DEV}.disable_ipv6=1" + +# NAT the guest out through the host's default interface. The guest needs +# real internet access for `docker pull` of the Lambda runtime image -- it's +# not just host<->guest traffic like the plain S3-only version of this demo. +NET_BASE="$(echo "$TAP_IP" | awk -F. '{print $1"."$2"."$3".0"}')/24" +HOST_IFACE="$(ip route show default 2>/dev/null | awk '/default/ {print $5; exit}')" +if [[ -n "$HOST_IFACE" ]]; then + echo "[run] enabling NAT via $HOST_IFACE so the guest can reach the internet" + sudo sysctl -qw net.ipv4.ip_forward=1 + sudo iptables -t nat -C POSTROUTING -s "$NET_BASE" -o "$HOST_IFACE" -j MASQUERADE 2>/dev/null \ + || sudo iptables -t nat -A POSTROUTING -s "$NET_BASE" -o "$HOST_IFACE" -j MASQUERADE + sudo iptables -C FORWARD -i "$TAP_DEV" -o "$HOST_IFACE" -j ACCEPT 2>/dev/null \ + || sudo iptables -A FORWARD -i "$TAP_DEV" -o "$HOST_IFACE" -j ACCEPT + sudo iptables -C FORWARD -i "$HOST_IFACE" -o "$TAP_DEV" -m state --state RELATED,ESTABLISHED -j ACCEPT 2>/dev/null \ + || sudo iptables -A FORWARD -i "$HOST_IFACE" -o "$TAP_DEV" -m state --state RELATED,ESTABLISHED -j ACCEPT +else + echo "[run] warning: no default route found on host; guest will have no internet access" >&2 +fi + +# Any KEY=VALUE on the kernel command line that systemd doesn't otherwise +# recognize is ignored unless it's spelled systemd.setenv=KEY=VALUE, in which +# case it becomes a manager-wide environment variable that every unit +# (including localstack.service) inherits. This is how LOCALSTACK_AUTH_TOKEN +# gets from the CI secret into the guest without baking it into the image. +AUTH_ENV="" +if [[ -n "${LOCALSTACK_AUTH_TOKEN:-}" ]]; then + AUTH_ENV=" systemd.setenv=LOCALSTACK_AUTH_TOKEN=${LOCALSTACK_AUTH_TOKEN}" +fi + +BOOT_ARGS="console=ttyS0 reboot=k panic=1 pci=off ip=${VM_IP}::${TAP_IP}:${VM_MASK}::eth0:off${AUTH_ENV}" + +cat > "$CONFIG" < "$LOG" 2>&1 < /dev/null & +disown +echo $! | sudo tee "$PIDFILE" >/dev/null + +SSH_KEY="$WORK_DIR/images/id_rsa" +SSH_OPTS=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o LogLevel=ERROR + -o BatchMode=yes -o ConnectTimeout=5 -i "$SSH_KEY" "root@${VM_IP}") + +# A cold `lstk` pull of the LocalStack image can legitimately take minutes, +# so the overall budget below stays generous -- but a broken docker.service +# or localstack.service (lstk itself failing to start, no auto-restart) +# reaches a terminal "failed" state within seconds of boot, so there is no +# reason to wait out the full budget for that case. Poll for it (after a +# short boot grace period) and bail immediately once either has failed. +is_service_broken() { + [[ -f "$SSH_KEY" ]] || return 1 + local states + states=$(ssh "${SSH_OPTS[@]}" 'systemctl is-active docker.service localstack.service' 2>/dev/null || echo "") + [[ "$states" == *failed* ]] +} + +echo "[run] waiting for LocalStack to become healthy at http://${VM_IP}:4566 ..." +for i in $(seq 1 90); do + if curl -fsS "http://${VM_IP}:4566/_localstack/health" >/dev/null 2>&1; then + echo "[run] LocalStack is up: http://${VM_IP}:4566" + exit 0 + fi + if (( i > 12 )) && (( i % 6 == 0 )) && is_service_broken; then + echo "[run] docker.service or localstack.service failed inside the guest; not waiting further. Run 'make diagnose' for details." >&2 + exit 1 + fi + sleep 5 +done + +echo "[run] timed out waiting for LocalStack; check $LOG or 'make diagnose'" >&2 +# One more data point while the VM is still up: is the container actually +# listening, and is it reachable from *inside* the guest (loopback) even +# though it is not reachable from us (over the tap network)? That tells us +# whether this is a bind-address issue or a NAT/forwarding issue. +if [[ -f "$SSH_KEY" ]]; then + echo "[run] --- port binding + in-guest reachability ---" >&2 + ssh "${SSH_OPTS[@]}" ' + echo "listening sockets on 4566:"; ss -tlnp | grep -E ":(4566)\b" || echo "(none)"; + echo "docker port mappings:"; docker ps --format "{{.Names}}: {{.Ports}}" 2>/dev/null || echo "(docker ps failed)"; + echo "curl from inside the guest:"; + curl -fsS -m 5 http://localhost:4566/_localstack/health && echo || echo "(failed)" + ' >&2 2>&1 || echo "[run] (SSH diagnostic itself failed)" >&2 +fi +exit 1 diff --git a/ls-on-firecracker/scripts/smoke-test.sh b/ls-on-firecracker/scripts/smoke-test.sh new file mode 100755 index 0000000..831e4a5 --- /dev/null +++ b/ls-on-firecracker/scripts/smoke-test.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Exercises the LocalStack instance running inside the microVM: +# - S3: create a bucket, put an object, read it back +# - Lambda: deploy a function that itself creates a bucket and lists all +# buckets, invoke it, and assert its response -- proving the Lambda's +# own AWS SDK calls land on the same LocalStack backend as the CLI calls +# above (it sees $BUCKET, and the bucket it creates is visible back here) +# Lambda execution happens via the guest's own Docker daemon, the same way +# it would against a normal `docker run localstack` setup. +set -euo pipefail + +: "${VM_IP:?run via 'make', not directly}" +: "${BUCKET:?}" +: "${LAMBDA_FN:?}" +: "${LAMBDA_BUCKET:?}" + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +export AWS_ACCESS_KEY_ID=test +export AWS_SECRET_ACCESS_KEY=test +export AWS_DEFAULT_REGION=us-east-1 +ENDPOINT="http://${VM_IP}:4566" +AWS="aws --endpoint-url $ENDPOINT" + +echo "[test] --- S3 ---" +echo "[test] creating bucket s3://${BUCKET}" +$AWS s3 mb "s3://${BUCKET}" + +TMP_FILE=$(mktemp) +echo "hello from a firecracker microVM" > "$TMP_FILE" +$AWS s3 cp "$TMP_FILE" "s3://${BUCKET}/hello.txt" >/dev/null +rm -f "$TMP_FILE" + +$AWS s3 ls "s3://${BUCKET}/" +echo "[test] S3 OK" + +echo "[test] --- Lambda ---" +ZIP_FILE=$(mktemp -u --suffix=.zip) +(cd "$SCRIPT_DIR/../fixtures" && zip -q "$ZIP_FILE" handler.py) + +echo "[test] creating function $LAMBDA_FN" +$AWS lambda create-function \ + --function-name "$LAMBDA_FN" \ + --runtime python3.12 \ + --handler handler.handler \ + --role arn:aws:iam::000000000000:role/lambda-role \ + --zip-file "fileb://${ZIP_FILE}" >/dev/null +rm -f "$ZIP_FILE" + +echo "[test] waiting for $LAMBDA_FN to become active (this pulls the Lambda runtime image)" +state="" +for _ in $(seq 1 40); do + state=$($AWS lambda get-function --function-name "$LAMBDA_FN" \ + --query 'Configuration.State' --output text 2>/dev/null || echo "") + [[ "$state" == "Active" ]] && break + sleep 5 +done +if [[ "$state" != "Active" ]]; then + echo "[test] FAILED: function never became active (last state: ${state:-unknown})" >&2 + exit 1 +fi + +echo "[test] invoking $LAMBDA_FN (it will create s3://${LAMBDA_BUCKET} and list all buckets)" +OUT_FILE=$(mktemp) +$AWS lambda invoke \ + --function-name "$LAMBDA_FN" \ + --cli-binary-format raw-in-base64-out \ + --payload "{\"name\":\"firecracker\",\"bucket\":\"${LAMBDA_BUCKET}\"}" \ + "$OUT_FILE" >/dev/null + +RESPONSE=$(cat "$OUT_FILE") +rm -f "$OUT_FILE" +echo "[test] response: $RESPONSE" + +MESSAGE=$(jq -r '.message' <<<"$RESPONSE") +if [[ "$MESSAGE" != "hello firecracker" ]]; then + echo "[test] FAILED: expected message 'hello firecracker', got '$MESSAGE'" >&2 + exit 1 +fi + +# The Lambda's own boto3 S3 calls must land on the same LocalStack backend +# as the CLI calls above: it should see the bucket created by this script +# ($BUCKET) and the one it just created itself ($LAMBDA_BUCKET). +mapfile -t LAMBDA_BUCKETS < <(jq -r '.buckets[]' <<<"$RESPONSE") +for expected in "$BUCKET" "$LAMBDA_BUCKET"; do + if [[ ! " ${LAMBDA_BUCKETS[*]} " == *" $expected "* ]]; then + echo "[test] FAILED: Lambda's bucket list did not include '$expected' (got: ${LAMBDA_BUCKETS[*]})" >&2 + exit 1 + fi +done + +# And the reverse: the bucket the Lambda created should be visible back here. +$AWS s3api head-bucket --bucket "$LAMBDA_BUCKET" + +echo "[test] Lambda OK (created s3://${LAMBDA_BUCKET}, saw both buckets, visible back on the CLI)" +echo "[test] all checks passed" diff --git a/ls-on-firecracker/scripts/teardown.sh b/ls-on-firecracker/scripts/teardown.sh new file mode 100755 index 0000000..9245e40 --- /dev/null +++ b/ls-on-firecracker/scripts/teardown.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# Stops the microVM and removes the tap device. Safe to run even if nothing +# is up. +set -euo pipefail + +: "${WORK_DIR:?run via 'make', not directly}" +: "${TAP_DEV:?}" +: "${TAP_IP:?}" + +PIDFILE="$WORK_DIR/firecracker.pid" + +if [[ -f "$PIDFILE" ]]; then + pid=$(cat "$PIDFILE") + if kill -0 "$pid" 2>/dev/null; then + echo "[down] stopping microVM (pid $pid)" + sudo kill "$pid" 2>/dev/null || true + sleep 1 + sudo kill -9 "$pid" 2>/dev/null || true + fi + rm -f "$PIDFILE" +fi +rm -f "$WORK_DIR/firecracker.sock" + +NET_BASE="$(echo "$TAP_IP" | awk -F. '{print $1"."$2"."$3".0"}')/24" +HOST_IFACE="$(ip route show default 2>/dev/null | awk '/default/ {print $5; exit}')" +if [[ -n "$HOST_IFACE" ]]; then + echo "[down] removing NAT rules for $NET_BASE via $HOST_IFACE" + sudo iptables -t nat -D POSTROUTING -s "$NET_BASE" -o "$HOST_IFACE" -j MASQUERADE 2>/dev/null || true + sudo iptables -D FORWARD -i "$TAP_DEV" -o "$HOST_IFACE" -j ACCEPT 2>/dev/null || true + sudo iptables -D FORWARD -i "$HOST_IFACE" -o "$TAP_DEV" -m state --state RELATED,ESTABLISHED -j ACCEPT 2>/dev/null || true +fi + +if ip link show "$TAP_DEV" &>/dev/null; then + echo "[down] removing tap device $TAP_DEV" + sudo ip link del "$TAP_DEV" 2>/dev/null || true +fi + +echo "[down] done"