From b38b526c9092b8871565576806e82f37a233a33c Mon Sep 17 00:00:00 2001 From: Shekhar Mudarapu Date: Fri, 31 Jul 2026 08:06:31 -0400 Subject: [PATCH 1/3] ci: gate the exec path on a real container, no KVM required MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The exec e2e test has existed since the exec dispatch fix but only ran by hand (`LANTERN_RUNTIME_E2E=1`), so the regression it guards was unprotected: for a long time `grpcSchedulerClient.Exec` returned a canned string with exit code 0, making a MISSING implementation indistinguishable from a command that ran and printed nothing. Mocks cannot catch that — only a real workload can. This runs on a stock GitHub runner. The runtime-manager's `docker` backend spawns a real container and hosted runners have Docker, so no nested virt is needed; that is why this can be an ordinary PR gate while the Firecracker boot job still waits on a KVM runner. The job builds the manager and scheduler, starts both, and runs the test against them. No Postgres service is needed: with DATABASE_URL unset the scheduler uses its in-memory store and runs always-leader, which is exactly a single-node test. The gate asserts the test RAN. `go test` reports a skipped test as a pass, so without this the job would go green while proving nothing — the same failure mode as the boot assertion that reported PASS for kernel-panicking VMs. The step fails on any `--- SKIP` and requires the PASS line for the real-VM test. Verified both directions locally before pushing: - env unset -> test SKIPs -> gate correctly FAILS - env set -> 2/2 subtests PASS against a live manager + scheduler -> gate passes, and the test's own cleanup left no containers behind Scheduler build command and workflow YAML both checked. --- .github/workflows/runtime-exec-e2e.yml | 144 +++++++++++++++++++++++++ 1 file changed, 144 insertions(+) create mode 100644 .github/workflows/runtime-exec-e2e.yml diff --git a/.github/workflows/runtime-exec-e2e.yml b/.github/workflows/runtime-exec-e2e.yml new file mode 100644 index 00000000..ff7a09fc --- /dev/null +++ b/.github/workflows/runtime-exec-e2e.yml @@ -0,0 +1,144 @@ +# runtime · exec e2e — live control-plane → scheduler → manager → container. +# +# Guards the exec dispatch path against the regression it was born from: for a +# long time `grpcSchedulerClient.Exec` returned a canned string with exit code +# 0, so a MISSING implementation was indistinguishable from a command that ran +# and printed nothing. Mocks cannot catch that — only a real workload can. +# +# NO KVM REQUIRED. The runtime-manager's `docker` backend spawns a real +# container, and GitHub's hosted runners have Docker. That is the whole reason +# this can be a normal PR gate while the Firecracker boot job +# (microvm-integration.yml) still needs a nested-virt runner. +name: "runtime · exec e2e (docker backend)" + +on: + pull_request: + paths: + - "services/control-plane/internal/handlers/runtime*.go" + - "services/runtime-manager/**" + - "services/runtime-scheduler/**" + - "packages/proto/lantern/v1/runtime.proto" + - ".github/workflows/runtime-exec-e2e.yml" + workflow_dispatch: {} + +permissions: + contents: read + +concurrency: + group: runtime-exec-e2e-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + exec-e2e: + name: exec-e2e + runs-on: ubuntu-latest + timeout-minutes: 25 + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Set up Go + uses: actions/setup-go@v5 + with: + go-version-file: services/control-plane/go.mod + + - name: Set up Rust (pinned by rust-toolchain.toml) + uses: dtolnay/rust-toolchain@stable + with: + toolchain: "1.93" + + - name: Cache cargo + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + services/runtime-manager/target + key: exec-e2e-cargo-${{ runner.os }}-${{ hashFiles('services/runtime-manager/Cargo.lock') }} + restore-keys: exec-e2e-cargo-${{ runner.os }}- + + # Debug build: this test exercises wiring, not performance, and a release + # build costs several minutes per run. + - name: Build runtime-manager + runtime-scheduler + run: | + set -euo pipefail + cargo build --manifest-path services/runtime-manager/Cargo.toml + (cd services/runtime-scheduler && go build -o bin/scheduler ./cmd/scheduler) + + # Pre-pull so the first Schedule is not also an image pull; a pull inside + # the test's own wait window is a flake source, not a signal. + - name: Pre-pull the workload image + run: docker pull python:3.11-slim + + - name: Start runtime-manager (docker backend) + run: | + set -euo pipefail + RUNTIME_BACKEND=docker \ + LISTEN_ADDR=0.0.0.0:50054 \ + AGENT_IMAGE=python:3.11-slim \ + LOG_LEVEL=info \ + ./services/runtime-manager/target/debug/lantern-runtime-manager \ + > /tmp/manager.log 2>&1 & + for _ in $(seq 1 60); do + (exec 3<>/dev/tcp/127.0.0.1/50054) 2>/dev/null && exec 3>&- 3<&- && break + sleep 1 + done + (exec 3<>/dev/tcp/127.0.0.1/50054) 2>/dev/null \ + || { echo "::error::runtime-manager never listened on :50054"; tail -50 /tmp/manager.log; exit 1; } + echo "runtime-manager up" + + - name: Start runtime-scheduler + run: | + set -euo pipefail + # DATABASE_URL intentionally unset: the scheduler then uses its + # in-memory store and runs always-leader, which is what a + # single-node test wants — no Postgres service needed. + LISTEN_ADDR=:50055 \ + HTTP_ADDR=:8085 \ + LANTERN_DEFAULT_MANAGER_ADDR=localhost:50054 \ + LOG_LEVEL=info \ + ./services/runtime-scheduler/bin/scheduler > /tmp/scheduler.log 2>&1 & + for _ in $(seq 1 60); do + (exec 3<>/dev/tcp/127.0.0.1/50055) 2>/dev/null && exec 3>&- 3<&- && break + sleep 1 + done + (exec 3<>/dev/tcp/127.0.0.1/50055) 2>/dev/null \ + || { echo "::error::runtime-scheduler never listened on :50055"; tail -50 /tmp/scheduler.log; exit 1; } + echo "runtime-scheduler up" + + # A skipped test reports as a pass to `go test`, which would make this + # gate green while proving nothing — the exact failure mode the boot + # assertion had. So require the subtests to have actually RUN. + - name: Run the exec e2e test + env: + LANTERN_RUNTIME_E2E: "1" + LANTERN_SCHEDULER_GRPC_ADDR: localhost:50055 + LANTERN_DEFAULT_MANAGER_ADDR: localhost:50054 + run: | + set -euo pipefail + cd services/control-plane + go test ./internal/handlers/ -run TestRuntimeExecE2E -v -count=1 -timeout 10m \ + 2>&1 | tee /tmp/exec-e2e.log + + if grep -q -- "--- SKIP" /tmp/exec-e2e.log; then + echo "::error::the exec e2e test SKIPPED — this gate must run it, not skip it" + exit 1 + fi + if ! grep -q -- "--- PASS: TestRuntimeExecE2E_RealVM" /tmp/exec-e2e.log; then + echo "::error::TestRuntimeExecE2E_RealVM did not report PASS" + exit 1 + fi + echo "exec e2e ran for real" + + - name: Service logs (on failure) + if: failure() + run: | + echo "----- runtime-manager -----"; tail -100 /tmp/manager.log || true + echo "----- runtime-scheduler ---"; tail -100 /tmp/scheduler.log || true + echo "----- containers ----------"; docker ps -a --filter "name=lantern-run" || true + + - name: Tear down + if: always() + run: | + docker ps -aq --filter "name=lantern-run" | xargs -r docker rm -f || true + pkill -f lantern-runtime-manager || true + pkill -f 'bin/scheduler' || true From 40c4ca50372d0d2a9a5afe9a2ef4b0c820a3a5f6 Mon Sep 17 00:00:00 2001 From: Shekhar Mudarapu Date: Fri, 31 Jul 2026 08:17:40 -0400 Subject: [PATCH 2/3] ci: install protoc for the exec e2e job MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The runtime-manager's build.rs compiles the runtime proto with prost-build, which shells out to protoc. Hosted runners do not ship it, so the build failed with 'Could not find protoc' — a dev Mac has it, which is exactly why this only appeared once the job ran in real CI rather than in local rehearsal. Same step sdlc-qa already uses. --- .github/workflows/runtime-exec-e2e.yml | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/.github/workflows/runtime-exec-e2e.yml b/.github/workflows/runtime-exec-e2e.yml index ff7a09fc..b99b2a0c 100644 --- a/.github/workflows/runtime-exec-e2e.yml +++ b/.github/workflows/runtime-exec-e2e.yml @@ -56,6 +56,13 @@ jobs: key: exec-e2e-cargo-${{ runner.os }}-${{ hashFiles('services/runtime-manager/Cargo.lock') }} restore-keys: exec-e2e-cargo-${{ runner.os }}- + # The runtime-manager's build.rs compiles the runtime proto with + # prost-build, which shells out to protoc. Hosted runners do not ship it + # (a dev Mac usually does, which is exactly why this only showed up in + # CI). Same step sdlc-qa uses. + - name: Install protoc + run: sudo apt-get update -qq && sudo apt-get install -y -qq protobuf-compiler + # Debug build: this test exercises wiring, not performance, and a release # build costs several minutes per run. - name: Build runtime-manager + runtime-scheduler From e501c2322a3e27a2c4c79168b30583fa0bd4a0b3 Mon Sep 17 00:00:00 2001 From: Shekhar Mudarapu Date: Fri, 31 Jul 2026 08:28:56 -0400 Subject: [PATCH 3/3] ci: register the manager as a node before running the exec e2e MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The test failed with 'placement failed: no nodes registered in cluster'. The manager self-registers with the scheduler via SCHEDULER_URL (POST /v1/nodes/heartbeat), and the workflow never set it — the LaunchAgent setup this was rehearsed against already had it, which is exactly why the gap only surfaced in real CI. Scheduler now starts FIRST so registration lands immediately, the manager gets SCHEDULER_URL and a stable NODE_NAME, and a readiness step waits for the node instead of racing it. That step watches the manager's own log rather than GET /v1/cluster: the endpoint requires an Authorization header, so polling it would 401 for the full timeout and report nothing useful. It also fails fast on 'heartbeat rejected' / 'heartbeat send failed' instead of waiting out the timeout. Verified locally that 'heartbeat ok' really is emitted under RUST_LOG=...scheduler_heartbeat=debug before relying on it. --- .github/workflows/runtime-exec-e2e.yml | 64 ++++++++++++++++++++------ 1 file changed, 50 insertions(+), 14 deletions(-) diff --git a/.github/workflows/runtime-exec-e2e.yml b/.github/workflows/runtime-exec-e2e.yml index b99b2a0c..bc02eb8c 100644 --- a/.github/workflows/runtime-exec-e2e.yml +++ b/.github/workflows/runtime-exec-e2e.yml @@ -76,10 +76,33 @@ jobs: - name: Pre-pull the workload image run: docker pull python:3.11-slim + - name: Start runtime-scheduler + run: | + set -euo pipefail + # DATABASE_URL intentionally unset: the scheduler then uses its + # in-memory store and runs always-leader, which is what a + # single-node test wants — no Postgres service needed. + LISTEN_ADDR=:50055 \ + HTTP_ADDR=:8085 \ + LANTERN_DEFAULT_MANAGER_ADDR=localhost:50054 \ + LOG_LEVEL=info \ + ./services/runtime-scheduler/bin/scheduler > /tmp/scheduler.log 2>&1 & + for _ in $(seq 1 60); do + (exec 3<>/dev/tcp/127.0.0.1/50055) 2>/dev/null && exec 3>&- 3<&- && break + sleep 1 + done + (exec 3<>/dev/tcp/127.0.0.1/50055) 2>/dev/null \ + || { echo "::error::runtime-scheduler never listened on :50055"; tail -50 /tmp/scheduler.log; exit 1; } + echo "runtime-scheduler up" + - name: Start runtime-manager (docker backend) run: | set -euo pipefail RUNTIME_BACKEND=docker \ + SCHEDULER_URL=http://localhost:8085 \ + NODE_NAME=ci-node \ + NODE_ADVERTISE_ADDR=localhost:50054 \ + RUST_LOG=info,lantern_runtime_manager::scheduler_heartbeat=debug \ LISTEN_ADDR=0.0.0.0:50054 \ AGENT_IMAGE=python:3.11-slim \ LOG_LEVEL=info \ @@ -93,24 +116,37 @@ jobs: || { echo "::error::runtime-manager never listened on :50054"; tail -50 /tmp/manager.log; exit 1; } echo "runtime-manager up" - - name: Start runtime-scheduler + # The manager self-registers with the scheduler (POST /v1/nodes/heartbeat) + # via SCHEDULER_URL. Placement fails with "no nodes registered in cluster" + # until that lands, so wait for the node rather than racing it — the + # LaunchAgent setup this was rehearsed against already had SCHEDULER_URL, + # which is why the gap only appeared in CI. + # The manager self-registers with the scheduler (POST /v1/nodes/heartbeat) + # via SCHEDULER_URL. Placement fails with "no nodes registered in cluster" + # until that lands, so wait for it rather than racing it — the LaunchAgent + # setup this was rehearsed against already had SCHEDULER_URL, which is why + # the gap only appeared in CI. + # + # Waits on the manager's own log, not GET /v1/cluster: that endpoint + # requires an Authorization header, so polling it would 401 forever and + # burn the timeout instead of reporting anything useful. + - name: Wait for the manager to register as a node run: | set -euo pipefail - # DATABASE_URL intentionally unset: the scheduler then uses its - # in-memory store and runs always-leader, which is what a - # single-node test wants — no Postgres service needed. - LISTEN_ADDR=:50055 \ - HTTP_ADDR=:8085 \ - LANTERN_DEFAULT_MANAGER_ADDR=localhost:50054 \ - LOG_LEVEL=info \ - ./services/runtime-scheduler/bin/scheduler > /tmp/scheduler.log 2>&1 & for _ in $(seq 1 60); do - (exec 3<>/dev/tcp/127.0.0.1/50055) 2>/dev/null && exec 3>&- 3<&- && break - sleep 1 + if grep -q "heartbeat ok" /tmp/manager.log 2>/dev/null; then + echo "node registered with the scheduler"; exit 0 + fi + if grep -qE "heartbeat (rejected|send failed)" /tmp/manager.log 2>/dev/null; then + echo "::error::the manager could not register with the scheduler" + tail -40 /tmp/manager.log; exit 1 + fi + sleep 2 done - (exec 3<>/dev/tcp/127.0.0.1/50055) 2>/dev/null \ - || { echo "::error::runtime-scheduler never listened on :50055"; tail -50 /tmp/scheduler.log; exit 1; } - echo "runtime-scheduler up" + echo "::error::manager never registered with the scheduler (timeout)" + echo "--- manager ---"; tail -40 /tmp/manager.log || true + echo "--- scheduler ---"; tail -40 /tmp/scheduler.log || true + exit 1 # A skipped test reports as a pass to `go test`, which would make this # gate green while proving nothing — the exact failure mode the boot