diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b645f4b8..4076d8f6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -37,6 +37,9 @@ jobs: - name: Sandbox-runner liveness checks run: tests/sandbox_runner_healthcheck.sh + - name: Sandbox-runner metrics discovery and network policy + run: tests/sandbox_runner_metrics.sh + - name: Bridge pairing rollout safety run: tests/bridge_pairing_rollout.sh @@ -121,6 +124,17 @@ jobs: - name: Bun tests run: bun run test + - name: File-heavy workspace cleanup on tmpfs + run: | + docker run --rm --user 0 \ + --tmpfs /tmp:rw,size=1g \ + --mount "type=bind,source=$GITHUB_WORKSPACE,target=/work,readonly" \ + --workdir /work/api \ + --env SANDBOX_CLEANUP_TMPFS_TEST=1 \ + --env SANDBOX_LOG_LEVEL=error \ + oven/bun:1.3.14-debian \ + bun test src/cleanup.integration.test.ts + service-unit-tests: name: Service Unit Tests runs-on: ubuntu-latest diff --git a/api/README.md b/api/README.md index 38e4cd98..34b32145 100644 --- a/api/README.md +++ b/api/README.md @@ -139,3 +139,70 @@ validation behavior. Roll out this sandbox behavior before enabling timeout forwarding in the service's plain `/exec` handler. Older sandboxes reject caps above their local runtime limit; older services remain compatible with updated sandboxes. + +### Runner memory after cleanup + +`Job.cleanup()` emits `Post-cleanup resources` after workspace removal and UID +release (or quarantine). `/execute` now waits for cleanup before sending success +or execution-failure responses. Artifact upload still finishes before cleanup. +Persistent session workspaces and their pinned UIDs are intentionally preserved; +failed disposable cleanup still quarantines the directory and retains its UID +for the existing retry path. This change does not alter those policies. + +The following metrics use only fixed, low-cardinality labels: + +| Metric suffix (prefix `codeapi_sandbox_`) | Meaning | +| --- | --- | +| `post_cleanup_memory_bytes{kind}` | `current`, `anon`, `file`, `shmem` at the visible cgroup v2 mount root | +| `post_cleanup_tmp_used{resource}` | `/tmp` allocated `bytes`, allocated `inodes`, and `is_tmpfs` (0 or 1) | +| `post_cleanup_workspaces{kind}` | Remaining `disposable`, `session`, and `other` entries in `/tmp/sandbox` | +| `cleanup_total{mode,outcome}` | Disposable/session cleanup attempts: `removed`, `preserved`, `retained`, `error` | +| `post_cleanup_sample_success{source}` | Whether `memory`, `tmp`, or `workspaces` was readable on the last sample | +| `post_cleanup_timestamp_seconds` | When the last cleanup sample was taken | + +These are runner-wide **last-cleanup samples**, not live gauges or per-job +memory attribution. Other jobs can still be running. Reaper-only changes are +visible on the next job cleanup, not immediately. Interpret the workspace counts +alongside active executions and the sample timestamp. Unavailable sources are +reported as unavailable and their old gauge values are removed, never replaced +with a healthy-looking zero. The structured log also includes active UID slots +and the number of retained cleanup retries. Synthetic jobs update metrics while +suppressing successful per-job logs as before. + +The visible cgroup root covers the API and sibling NsJail cgroups. The old +`Post-execution memory` log instead samples `/proc/self/cgroup` **before** job +cleanup, so it can have a different scope. `file` includes `shmem`; do not add +them or assume all `file` memory is reclaimable page cache. + +`statfs` measures allocation, including deleted-but-open files on the sampled +mount, without walking user files. The runner's 1 GiB `/tmp` mount is distinct +from each jail's 20 MiB `/tmp` mount. Its ceiling does not cap all container +memory. If workspace counts fall but shmem does not, investigate descendant +processes and retained mounts as well as files. A stable API-process FD count +alone cannot exclude those cases. Memory requests and HPA settings are unchanged. + +#### Cleanup stress tests + +Ordinary unit tests cover metrics, unavailable sources, session/disposable +classification and response ordering. The opt-in stress tests must run in an +**isolated test container**, never on an active runner: + +```bash +# From api/, with /tmp mounted as tmpfs and per-job chown available: +SANDBOX_CLEANUP_TMPFS_TEST=1 bun test src/cleanup.integration.test.ts + +# Additionally requires the runner's NsJail binary, spec-guard, config, +# and normal namespace/cgroup permissions. Run this file alone. +NSJAIL_CONFIG=/sandbox_api/config/sandbox.cfg \ +SANDBOX_CLEANUP_NSJAIL_TEST=1 bun test src/cleanup.integration.test.ts +``` + +The first test repeats real Job priming and cleanup with large payloads and many +small files, checking tmpfs bytes/inodes and UID slots after every iteration. +CI runs it in a dedicated Bun container with a 1 GiB `/tmp` tmpfs. It does not +execute NsJail. The second test uses real NsJail jobs covering normal completion, +timeout and output overflow, including a detached child holding an unlinked file. +It asserts no processes remain under the job UID before removing its workspace, +then checks workspace removal and post-cleanup tmpfs allocation. This test is +skipped unless explicitly enabled; a unit-test pass is not proof of namespace +teardown on the production kernel. diff --git a/api/src/api/v2-cleanup.test.ts b/api/src/api/v2-cleanup.test.ts new file mode 100644 index 00000000..d1efee5d --- /dev/null +++ b/api/src/api/v2-cleanup.test.ts @@ -0,0 +1,189 @@ +import { afterAll, afterEach, beforeAll, expect, test } from 'bun:test'; +import express from 'express'; +import { mkdtemp, rm, writeFile } from 'fs/promises'; +import type { Server } from 'http'; +import { tmpdir } from 'os'; +import { join } from 'path'; +import { config } from '../config'; +import { Job } from '../job'; +import { loadPackage } from '../runtime'; +import { ValidationError } from '../validation'; +import router from './v2'; + +let server: Server; +let url: string; +let directory: string; +const language = 'cleanup-order-test'; +const originalPrime = Job.prototype.prime; +const originalExecute = Job.prototype.execute; +const originalUpload = Job.prototype.uploadGeneratedFiles; +const originalCleanup = Job.prototype.cleanup; +const requireManifest = config.require_execution_manifest; + +beforeAll(async () => { + directory = await mkdtemp(join(tmpdir(), 'cleanup-order-')); + await writeFile( + join(directory, 'pkg-info.json'), + JSON.stringify({ language, version: '1.0.0', aliases: [] }) + ); + loadPackage(directory); + const app = express(); + app.use(router); + await new Promise(resolve => { + server = app.listen(0, '127.0.0.1', () => resolve()); + }); + const address = server.address(); + url = `http://127.0.0.1:${ + typeof address === 'object' && address ? address.port : 0 + }/execute`; +}); +afterAll(async () => { + await new Promise(resolve => server.close(() => resolve())); + await rm(directory, { recursive: true, force: true }); +}); +afterEach(() => { + Job.prototype.prime = originalPrime; + Job.prototype.execute = originalExecute; + Job.prototype.uploadGeneratedFiles = originalUpload; + Job.prototype.cleanup = originalCleanup; + config.require_execution_manifest = requireManifest; +}); + +for (const outcome of [ + 'success', + 'prime_failure', + 'execution_failure', + 'validation_failure', +] as const) { + test(`${outcome} waits for cleanup before sending a response and cleans exactly once`, async () => { + config.require_execution_manifest = false; + const events: string[] = []; + let releaseCleanup!: () => void; + const cleanupGate = new Promise(resolve => { + releaseCleanup = resolve; + }); + let markCleanupStarted!: () => void; + const cleanupStarted = new Promise(resolve => { + markCleanupStarted = resolve; + }); + Job.prototype.prime = async function () { + events.push('prime'); + if (outcome === 'prime_failure') + throw new Error('scripted prime failure'); + }; + Job.prototype.execute = async function () { + events.push('execute'); + if (outcome === 'execution_failure') + throw new Error('scripted execution failure'); + if (outcome === 'validation_failure') + throw new ValidationError('scripted validation failure'); + return { files: [{ id: 'artifact', name: 'result.txt' }] } as Awaited< + ReturnType + >; + }; + Job.prototype.uploadGeneratedFiles = async function () { + events.push('upload'); + return new Set(['artifact']); + }; + Job.prototype.cleanup = async function () { + events.push('cleanup-start'); + markCleanupStarted(); + await cleanupGate; + events.push('cleanup-end'); + }; + let responseArrived = false; + const pending = fetch(url, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + language, + version: '1.0.0', + files: [{ name: 'main.txt', content: 'test' }], + }), + }).then(response => { + responseArrived = true; + return response; + }); + try { + await Promise.race([ + cleanupStarted, + pending.then(response => { + throw new Error(`Response preceded cleanup: ${response.status}`); + }), + ]); + // Give an incorrectly early res.json() time to reach the client. + await new Promise(resolve => setTimeout(resolve, 30)); + expect(responseArrived).toBe(false); + } finally { + releaseCleanup(); + const response = await pending; + expect(response.status).toBe( + outcome === 'success' + ? 200 + : outcome === 'validation_failure' + ? 400 + : 500 + ); + await response.text(); + } + expect(events.filter(event => event === 'cleanup-start')).toHaveLength(1); + expect(events[events.length - 1]).toBe('cleanup-end'); + if (outcome === 'success') + expect(events).toEqual([ + 'prime', + 'execute', + 'upload', + 'cleanup-start', + 'cleanup-end', + ]); + }); +} + +test('client disconnect does not clean a workspace while execution is still running', async () => { + config.require_execution_manifest = false; + let releaseExecution!: () => void; + const executionGate = new Promise(resolve => { + releaseExecution = resolve; + }); + let markExecuting!: () => void; + const executing = new Promise(resolve => { + markExecuting = resolve; + }); + let markCleaned!: () => void; + const cleaned = new Promise(resolve => { + markCleaned = resolve; + }); + let cleanupCount = 0; + Job.prototype.prime = async function () {}; + Job.prototype.execute = async function () { + markExecuting(); + await executionGate; + return {} as Awaited>; + }; + Job.prototype.cleanup = async function () { + cleanupCount++; + markCleaned(); + }; + const controller = new AbortController(); + const pending = fetch(url, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + signal: controller.signal, + body: JSON.stringify({ + language, + version: '1.0.0', + files: [{ name: 'main.txt', content: 'test' }], + }), + }).catch(() => undefined); + try { + await executing; + controller.abort(); + await pending; + await new Promise(resolve => setTimeout(resolve, 30)); + expect(cleanupCount).toBe(0); + } finally { + releaseExecution(); + await cleaned; + } + expect(cleanupCount).toBe(1); +}); diff --git a/api/src/api/v2.ts b/api/src/api/v2.ts index e2f252ce..121cf2e3 100644 --- a/api/src/api/v2.ts +++ b/api/src/api/v2.ts @@ -588,9 +588,13 @@ router.post('/execute', express.json({ limit: config.execute_body_limit }), asyn } } + /* Upload must finish before cleanup, and cleanup must settle before + * acknowledging completion. Failed removals retain a quarantined UID. */ + await cleanupHandler(); metricsOutcome = 'success'; return res.status(200).json(result); } catch (error) { + await cleanupHandler(); /* Deliberately BEFORE the ValidationError branch below: once priming has * completed, the workspace has been written to, so any later failure — * including a validation one — leaves state the next execute must not diff --git a/api/src/cleanup-metrics.test.ts b/api/src/cleanup-metrics.test.ts new file mode 100644 index 00000000..e1f61b22 --- /dev/null +++ b/api/src/cleanup-metrics.test.ts @@ -0,0 +1,110 @@ +import { afterEach, describe, expect, test } from 'bun:test'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { + postCleanupMemory, + postCleanupTmp, + postCleanupWorkspaces, + readCleanupResources, + recordCleanupResources, +} from './cleanup-metrics'; + +const roots: string[] = []; +afterEach(() => { + for (const root of roots.splice(0)) + fs.rmSync(root, { recursive: true, force: true }); +}); + +function fixture() { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'cleanup-metrics-')); + roots.push(root); + const cgroupRoot = path.join(root, 'cgroup'); + const workspaceRoot = path.join(root, 'workspaces'); + fs.mkdirSync(cgroupRoot); + fs.mkdirSync(workspaceRoot); + fs.writeFileSync(path.join(cgroupRoot, 'memory.current'), '1048576\n'); + fs.writeFileSync( + path.join(cgroupRoot, 'memory.stat'), + 'anon 262144\nfile 786432\nshmem 524288\n' + ); + return { cgroupRoot, workspaceRoot, tmpRoot: root }; +} + +describe('post-cleanup resources', () => { + test('reads byte-accurate cgroup counters without adding shmem to file', () => { + const paths = fixture(); + // The API child cgroup excludes sibling jails. Sample the visible root. + fs.mkdirSync(path.join(paths.cgroupRoot, 'sandbox_api')); + fs.writeFileSync( + path.join(paths.cgroupRoot, 'sandbox_api/memory.current'), + '1' + ); + const snapshot = readCleanupResources(paths); + expect(snapshot.memory).toEqual({ + current: 1048576, + anon: 262144, + file: 786432, + shmem: 524288, + }); + expect(snapshot.tmp?.bytes).toBeGreaterThanOrEqual(0); + expect(snapshot.tmp?.inodes).toBeGreaterThanOrEqual(0); + expect(snapshot.errors).toEqual({}); + }); + + test('separates disposable, persistent and unexpected entries without following symlinks', () => { + const paths = fixture(); + fs.mkdirSync(path.join(paths.workspaceRoot, 'ws_active')); + fs.mkdirSync(path.join(paths.workspaceRoot, 'session')); + fs.mkdirSync(path.join(paths.workspaceRoot, 'unknown')); + fs.symlinkSync(paths.tmpRoot, path.join(paths.workspaceRoot, 'ws_symlink')); + expect(readCleanupResources(paths).workspaces).toEqual({ + disposable: 1, + session: 1, + other: 2, + }); + }); + + test('does not turn missing or malformed statistics into a healthy zero', () => { + const paths = fixture(); + fs.writeFileSync( + path.join(paths.cgroupRoot, 'memory.stat'), + 'anon 1\nfile 2\n' + ); + expect(readCleanupResources(paths).memory).toBeNull(); + fs.writeFileSync( + path.join(paths.cgroupRoot, 'memory.current'), + 'not-a-number' + ); + const snapshot = readCleanupResources(paths); + expect(snapshot.memory).toBeNull(); + expect(snapshot.tmp).not.toBeNull(); + expect(snapshot.workspaces).not.toBeNull(); + expect(snapshot.errors.memory).toBe('unavailable'); + }); + + test('unreadable sources remove stale gauges without failing cleanup', async () => { + const paths = fixture(); + const context = { + job: 'test', + mode: 'disposable', + outcome: 'removed', + suppressSuccessLogs: true, + } as const; + recordCleanupResources(context, readCleanupResources(paths)); + expect((await postCleanupMemory.get()).values).toHaveLength(4); + expect((await postCleanupTmp.get()).values).toHaveLength(3); + expect((await postCleanupWorkspaces.get()).values).toHaveLength(3); + fs.rmSync(paths.tmpRoot, { recursive: true }); + const snapshot = readCleanupResources(paths); + expect(snapshot.errors).toEqual({ + memory: 'ENOENT', + tmp: 'ENOENT', + workspaces: 'ENOENT', + }); + expect(() => recordCleanupResources(context, snapshot)).not.toThrow(); + expect((await postCleanupMemory.get()).values).toHaveLength(0); + expect((await postCleanupTmp.get()).values).toHaveLength(0); + expect((await postCleanupWorkspaces.get()).values).toHaveLength(0); + }); +}); diff --git a/api/src/cleanup-metrics.ts b/api/src/cleanup-metrics.ts new file mode 100644 index 00000000..7444960d --- /dev/null +++ b/api/src/cleanup-metrics.ts @@ -0,0 +1,185 @@ +import * as fs from 'fs'; +import * as path from 'path'; +import { Counter, Gauge } from 'prom-client'; +import { logger } from './logger'; +import { + SANDBOX_WORKSPACE_ROOT, + SESSION_WORKSPACE_ID, + WORKSPACE_ID_PREFIX, + retainedWorkspaceCleanupCount, + sandboxJobUidPool, +} from './workspace-isolation'; + +const memoryKinds = ['current', 'anon', 'file', 'shmem'] as const; +const workspaceKinds = ['disposable', 'session', 'other'] as const; +type MemoryKind = typeof memoryKinds[number]; +type WorkspaceKind = typeof workspaceKinds[number]; + +export const postCleanupMemory = new Gauge({ + name: 'codeapi_sandbox_post_cleanup_memory_bytes', + help: 'Last post-cleanup cgroup-root memory sample; file includes shmem, not additive', + labelNames: ['kind'] as const, +}); +export const postCleanupTmp = new Gauge({ + name: 'codeapi_sandbox_post_cleanup_tmp_used', + help: 'Last post-cleanup /tmp filesystem allocation, including unlinked open files', + labelNames: ['resource'] as const, +}); +export const postCleanupWorkspaces = new Gauge({ + name: 'codeapi_sandbox_post_cleanup_workspaces', + help: 'Remaining workspace-root entries after cleanup; includes other active jobs', + labelNames: ['kind'] as const, +}); +const cleanupAttempts = new Counter({ + name: 'codeapi_sandbox_cleanup_total', + help: 'Job cleanup attempts; session workspaces are intentionally preserved', + labelNames: ['mode', 'outcome'] as const, +}); +const sampleSuccess = new Gauge({ + name: 'codeapi_sandbox_post_cleanup_sample_success', + help: 'Whether each source was readable on the last post-cleanup sample', + labelNames: ['source'] as const, +}); +const sampleTimestamp = new Gauge({ + name: 'codeapi_sandbox_post_cleanup_timestamp_seconds', + help: 'Timestamp of the last job cleanup resource sample, not a live usage gauge', +}); + +export interface CleanupResourceSnapshot { + memory: Record | null; + tmp: { bytes: number; inodes: number; isTmpfs: boolean } | null; + workspaces: Record | null; + errors: Partial>; +} + +/** Read only fixed kernel files and the workspace root, never recurse into + * untrusted workspace trees or follow their symlinks. Synchronous sampling + * avoids older, slower samples overwriting newer ones during parallel cleanup. + * The visible cgroup mount root includes the API AND sibling NsJail cgroups; + * /proc/self/cgroup may point only at the delegated sandbox_api child instead. + * These are runner-wide observations, not memory attributable to one job. */ +export function readCleanupResources( + paths: { + cgroupRoot?: string; + tmpRoot?: string; + workspaceRoot?: string; + } = {} +): CleanupResourceSnapshot { + const snapshot: CleanupResourceSnapshot = { + memory: null, + tmp: null, + workspaces: null, + errors: {}, + }; + function sample( + source: keyof CleanupResourceSnapshot['errors'], + read: () => T + ): T | null { + try { + return read(); + } catch (error) { + snapshot.errors[source] = + (error as NodeJS.ErrnoException).code ?? 'unavailable'; + return null; + } + } + snapshot.memory = sample('memory', () => { + const root = paths.cgroupRoot ?? '/sys/fs/cgroup'; + const current = fs + .readFileSync(path.join(root, 'memory.current'), 'utf8') + .trim(); + const stat = fs.readFileSync(path.join(root, 'memory.stat'), 'utf8'); + const values = Object.fromEntries( + stat + .trim() + .split('\n') + .map(line => line.trim().split(/\s+/)) + ); + const result = {} as Record; + for (const kind of memoryKinds) { + const raw = kind === 'current' ? current : values[kind]; + if (!raw || !/^\d+$/.test(raw) || !Number.isSafeInteger(Number(raw))) { + throw new Error('Invalid memory statistic'); + } + result[kind] = Number(raw); + } + return result; + }); + snapshot.tmp = sample('tmp', () => { + const stat = fs.statfsSync(paths.tmpRoot ?? '/tmp'); + return { + bytes: (stat.blocks - stat.bfree) * stat.bsize, + inodes: stat.files - stat.ffree, + isTmpfs: stat.type === 0x01021994, + }; + }); + snapshot.workspaces = sample('workspaces', () => { + const counts = { disposable: 0, session: 0, other: 0 }; + for (const entry of fs.readdirSync( + paths.workspaceRoot ?? SANDBOX_WORKSPACE_ROOT, + { withFileTypes: true } + )) { + const kind = + entry.isDirectory() && entry.name === SESSION_WORKSPACE_ID + ? 'session' + : entry.isDirectory() && entry.name.startsWith(WORKSPACE_ID_PREFIX) + ? 'disposable' + : 'other'; + counts[kind]++; + } + return counts; + }); + return snapshot; +} + +export function recordCleanupResources( + context: { + job: string; + mode: 'disposable' | 'session'; + outcome: 'removed' | 'preserved' | 'retained' | 'error'; + suppressSuccessLogs?: boolean; + }, + snapshot = readCleanupResources() +): void { + cleanupAttempts.inc({ mode: context.mode, outcome: context.outcome }); + sampleTimestamp.set(Date.now() / 1000); + for (const source of ['memory', 'tmp', 'workspaces'] as const) { + sampleSuccess.set({ source }, snapshot[source] === null ? 0 : 1); + } + for (const kind of memoryKinds) { + if (snapshot.memory) postCleanupMemory.set({ kind }, snapshot.memory[kind]); + else postCleanupMemory.remove({ kind }); + } + for (const resource of ['bytes', 'inodes', 'is_tmpfs'] as const) { + if (snapshot.tmp) { + postCleanupTmp.set( + { resource }, + resource === 'is_tmpfs' + ? Number(snapshot.tmp.isTmpfs) + : snapshot.tmp[resource] + ); + } else postCleanupTmp.remove({ resource }); + } + for (const kind of workspaceKinds) { + if (snapshot.workspaces) + postCleanupWorkspaces.set({ kind }, snapshot.workspaces[kind]); + else postCleanupWorkspaces.remove({ kind }); + } + if ( + !context.suppressSuccessLogs || + context.outcome === 'retained' || + context.outcome === 'error' + ) { + logger.info( + { + job: context.job, + mode: context.mode, + outcome: context.outcome, + ...snapshot, + activeUidSlots: sandboxJobUidPool.activeCount(), + retainedCleanups: retainedWorkspaceCleanupCount(), + }, + 'Post-cleanup resources' + ); + } +} diff --git a/api/src/cleanup.integration.test.ts b/api/src/cleanup.integration.test.ts new file mode 100644 index 00000000..0d08e804 --- /dev/null +++ b/api/src/cleanup.integration.test.ts @@ -0,0 +1,184 @@ +import { expect, test } from 'bun:test'; +import * as fs from 'fs'; +import * as fsp from 'fs/promises'; +import * as os from 'os'; +import * as path from 'path'; +import * as semver from 'semver'; +import { config } from './config'; +import { Job } from './job'; +import type { Runtime } from './runtime'; +import { readCleanupResources } from './cleanup-metrics'; +import { prepareWorkspaceRoot, sandboxJobUidPool } from './workspace-isolation'; + +// Run only in a dedicated Linux test container with /tmp mounted as tmpfs. +// The real-jail variant additionally needs the runner image's namespace/cgroup +// permissions, nsjail, spec-guard and NSJAIL_CONFIG. Never target a live runner. +const tmpfsEnabled = process.env.SANDBOX_CLEANUP_TMPFS_TEST === '1'; +const nsjailEnabled = process.env.SANDBOX_CLEANUP_NSJAIL_TEST === '1'; + +function runtime(pkgdir: string): Runtime { + return { + language: 'bash', + version: new semver.SemVer('5.2.0'), + aliases: [], + pkgdir, + compiled: false, + env_vars: {}, + timeouts: { compile: 1000, run: 1000 }, + cpu_times: { compile: 1000, run: 1000 }, + memory_limits: { compile: 128 * 1048576, run: 128 * 1048576 }, + max_process_count: 64, + max_open_files: 64, + max_file_size: 16 * 1048576, + output_max_size: 4096, + }; +} + +function jobFor(rt: Runtime, content: string): Job { + return new Job({ + session_id: 'cleanup-stress', + runtime: rt, + files: [{ name: 'main.sh', content }], + args: [], + stdin: '', + timeouts: rt.timeouts, + cpu_times: rt.cpu_times, + memory_limits: rt.memory_limits, + is_synthetic: true, + }); +} + +function workspace(job: Job): string { + return (job as unknown as { submissionDir: string }).submissionDir; +} + +function assertTmpfsBaseline( + baseline: ReturnType +): void { + const after = readCleanupResources(); + expect(after.tmp?.isTmpfs).toBe(true); + // Allow a few metadata pages, not one retained 8 MiB payload per execution. + expect(after.tmp!.bytes).toBeLessThanOrEqual(baseline.tmp!.bytes + 64 * 1024); + expect(after.tmp!.inodes).toBeLessThanOrEqual(baseline.tmp!.inodes + 4); + expect(after.workspaces).toEqual(baseline.workspaces); +} + +test.skipIf(!tmpfsEnabled)( + 'repeated file-heavy Job cleanup returns tmpfs bytes and inodes near baseline', + async () => { + expect(process.platform).toBe('linux'); + await prepareWorkspaceRoot(); + const baseline = readCleanupResources(); + expect(baseline.tmp?.isTmpfs).toBe(true); + const activeBefore = sandboxJobUidPool.activeCount(); + const payload = Buffer.alloc(8 * 1048576, 0x61); + for (let iteration = 0; iteration < 24; iteration++) { + const job = jobFor(runtime('/tmp'), 'echo test'); + let dir = ''; + try { + await job.prime(); + dir = workspace(job); + await fsp.writeFile(path.join(dir, 'payload'), payload); + for (let i = 0; i < 64; i++) + await fsp.writeFile(path.join(dir, `small-${i}`), 'x'); + expect(readCleanupResources().tmp!.bytes).toBeGreaterThan( + baseline.tmp!.bytes + payload.length + ); + } finally { + await job.cleanup(); + } + expect(fs.existsSync(dir)).toBe(false); + expect(sandboxJobUidPool.activeCount()).toBe(activeBefore); + assertTmpfsBaseline(baseline); + } + }, + 30_000 +); + +function processesForUid(uid: number): string[] { + return fs.readdirSync('/proc').filter(pid => { + if (!/^\d+$/.test(pid)) return false; + try { + return ( + fs + .readFileSync(`/proc/${pid}/status`, 'utf8') + .match(/^Uid:\s+(\d+)/m)?.[1] === String(uid) + ); + } catch (error) { + if ( + (error as NodeJS.ErrnoException).code === 'ENOENT' || + (error as NodeJS.ErrnoException).code === 'ESRCH' + ) + return false; + throw error; + } + }); +} + +test.skipIf(!nsjailEnabled)( + 'real NsJail releases detached descendants and file-heavy workspaces on success, timeout and overflow', + async () => { + expect(process.platform).toBe('linux'); + expect(config.per_job_uids).toBe(true); + expect(process.getuid?.()).toBe(0); + await prepareWorkspaceRoot(); + const pkgdir = await fsp.mkdtemp( + path.join(os.tmpdir(), 'cleanup-runtime-') + ); + await fsp.chmod(pkgdir, 0o755); + await fsp.writeFile(path.join(pkgdir, 'run'), 'exec /bin/bash "$@"\n', { + mode: 0o644, + }); + const baseline = readCleanupResources(); + expect(baseline.tmp?.isTmpfs).toBe(true); + const activeBefore = sandboxJobUidPool.activeCount(); + try { + for (let iteration = 0; iteration < 12; iteration++) { + const ending = iteration % 3; + const job = jobFor( + runtime(pkgdir), + `set -eu +# Both linked workspace data and an unlinked open file held by a detached child. +dd if=/dev/zero of=payload bs=1048576 count=8 status=none +# Per-jail /tmp is a different mount from the runner's /tmp. +dd if=/dev/zero of=/tmp/held bs=1048576 count=8 status=none +exec 3 /mnt/data/child-ready; exec sleep 30' >/dev/null 2>&1 & +while [ ! -f child-ready ]; do sleep 0.01; done +${ + ending === 0 + ? 'echo complete' + : ending === 1 + ? 'sleep 30' + : "while :; do printf '%1024s' x; done" +} +` + ); + let dir = ''; + let uid = -1; + try { + await job.prime(); + dir = workspace(job); + uid = (job as unknown as { jobIdentity: { uid: number } }).jobIdentity + .uid; + const result = await job.execute(); + expect(result.run?.status).toBe( + ending === 0 ? null : ending === 1 ? 'TO' : 'OL' + ); + if (ending === 0) expect(result.run?.stdout).toContain('complete'); + // Check before workspace removal can conceal a surviving writer. + expect(processesForUid(uid)).toEqual([]); + } finally { + await job.cleanup(); + } + expect(fs.existsSync(dir)).toBe(false); + expect(sandboxJobUidPool.activeCount()).toBe(activeBefore); + assertTmpfsBaseline(baseline); + } + } finally { + await fsp.rm(pkgdir, { recursive: true, force: true }); + } + }, + 60_000 +); diff --git a/api/src/job-cleanup-metrics.test.ts b/api/src/job-cleanup-metrics.test.ts new file mode 100644 index 00000000..6806f637 --- /dev/null +++ b/api/src/job-cleanup-metrics.test.ts @@ -0,0 +1,118 @@ +import { afterEach, expect, spyOn, test } from 'bun:test'; +import * as fsp from 'fs/promises'; +import * as os from 'os'; +import * as path from 'path'; +import * as semver from 'semver'; +import { register } from 'prom-client'; +import { Job } from './job'; +import type { Runtime } from './runtime'; +import type { SessionWorkspace } from './session-workspace'; +import { + clearRetainedWorkspaceCleanupsForTest, + retainedWorkspaceCleanupCount, + retryRetainedWorkspaceCleanups, + sandboxJobUidPool, +} from './workspace-isolation'; + +import type { + SandboxJobIdentity, + SandboxWorkspaceLease, +} from './workspace-isolation'; + +interface Internals { + submissionDir: string; + workspaceLease?: SandboxWorkspaceLease; + jobIdentity?: SandboxJobIdentity; +} +const roots: string[] = []; +afterEach(async () => { + await retryRetainedWorkspaceCleanups(); + clearRetainedWorkspaceCleanupsForTest(); + for (const root of roots.splice(0)) + await fsp.rm(root, { recursive: true, force: true }); +}); + +async function fixture(persistent = false) { + const dir = await fsp.mkdtemp(path.join(os.tmpdir(), 'job-cleanup-metric-')); + roots.push(dir); + const identity = sandboxJobUidPool.acquire()!; + expect(identity).not.toBeNull(); + const lease = { dir, workspaceId: path.basename(dir), identity }; + const rt: Runtime = { + language: 'bash', + version: new semver.SemVer('5.2.0'), + aliases: [], + pkgdir: '/tmp', + compiled: false, + env_vars: {}, + timeouts: { compile: 1000, run: 1000 }, + cpu_times: { compile: 1000, run: 1000 }, + memory_limits: { compile: -1, run: -1 }, + max_process_count: 64, + max_open_files: 64, + max_file_size: 1000, + output_max_size: 1000, + }; + const job = new Job({ + runtime: rt, + files: [], + args: [], + stdin: '', + timeouts: rt.timeouts, + cpu_times: rt.cpu_times, + memory_limits: rt.memory_limits, + is_synthetic: true, + session: persistent ? ({} as SessionWorkspace) : undefined, + }); + Object.assign(job as unknown as Internals, { + submissionDir: dir, + workspaceLease: lease, + jobIdentity: identity, + }); + return { job, dir, identity }; +} + +async function counter(mode: string, outcome: string): Promise { + const metric = register.getSingleMetric('codeapi_sandbox_cleanup_total')!; + return ( + (await metric.get()).values.find( + value => value.labels.mode === mode && value.labels.outcome === outcome + )?.value ?? 0 + ); +} + +test('failed removal is measured as retained and does not release its UID before retry', async () => { + const activeBefore = sandboxJobUidPool.activeCount(); + const retainedBefore = await counter('disposable', 'retained'); + const { job, dir } = await fixture(); + const rm = spyOn(fsp, 'rm').mockRejectedValueOnce( + Object.assign(new Error('test busy workspace'), { code: 'EBUSY' }) + ); + try { + await job.cleanup(); + expect(await counter('disposable', 'retained')).toBe(retainedBefore + 1); + expect(sandboxJobUidPool.activeCount()).toBe(activeBefore + 1); + expect(retainedWorkspaceCleanupCount()).toBe(1); + expect((await fsp.lstat(dir)).isDirectory()).toBe(true); + } finally { + rm.mockRestore(); + await retryRetainedWorkspaceCleanups(); + } + expect(sandboxJobUidPool.activeCount()).toBe(activeBefore); + expect(await fsp.lstat(dir).catch(() => null)).toBeNull(); +}); + +test('session cleanup is measured as preserved and retains both files and pinned UID', async () => { + const activeBefore = sandboxJobUidPool.activeCount(); + const preservedBefore = await counter('session', 'preserved'); + const { job, dir, identity } = await fixture(true); + try { + await fsp.writeFile(path.join(dir, 'state'), 'keep'); + await job.cleanup(); + expect(await counter('session', 'preserved')).toBe(preservedBefore + 1); + expect(await fsp.readFile(path.join(dir, 'state'), 'utf8')).toBe('keep'); + expect(sandboxJobUidPool.activeCount()).toBe(activeBefore + 1); + } finally { + sandboxJobUidPool.release(identity); + } +}); diff --git a/api/src/job.ts b/api/src/job.ts index 0939de08..5c2fddef 100644 --- a/api/src/job.ts +++ b/api/src/job.ts @@ -14,6 +14,7 @@ import { logger as rootLogger } from './logger'; import { getRuntimes } from './runtime'; import { execute } from './nsjail'; import { config } from './config'; +import { recordCleanupResources } from './cleanup-metrics'; import { internalServiceHeaders } from './internal-service-auth'; import { EGRESS_GRANT_HEADER, EGRESS_ERROR_CODE_HEADER } from './egress'; import { injectTraceHeaders } from './telemetry'; @@ -2716,49 +2717,61 @@ export class Job { this.log.info('Cleaning up'); } - /* Session mode: the workspace and pinned UID belong to the long-lived - * session, not this job. Keep both so the next call sees prior files; - * teardown happens on the /terminate hook (or explicit session reset). */ - if (this.session) { - this.workspaceLease = undefined; - this.submissionDir = ''; - this.jobIdentity = undefined; - return; - } - - let workspaceRemoved = true; - const workspaceLease = this.workspaceLease; - const jobIdentity = this.jobIdentity; - - if (workspaceLease) { - try { - workspaceRemoved = await cleanupSandboxWorkspace(workspaceLease); - } catch (error) { - workspaceRemoved = false; - this.log.error({ submissionDir: this.submissionDir, err: error }, 'Failed to clean up'); - } finally { + let outcome: 'removed' | 'preserved' | 'retained' | 'error' = 'error'; + try { + /* Session mode: the workspace and pinned UID belong to the long-lived + * session, not this job. Keep both so the next call sees prior files; + * teardown happens on the /terminate hook (or explicit session reset). */ + if (this.session) { this.workspaceLease = undefined; this.submissionDir = ''; + this.jobIdentity = undefined; + outcome = 'preserved'; + return; } - } - if (jobIdentity) { - if (!workspaceLease || workspaceRemoved) { - releaseJobIdentity(jobIdentity); - } else { - retainWorkspaceCleanupUntilRemoved(workspaceLease, () => { + let workspaceRemoved = true; + const workspaceLease = this.workspaceLease; + const jobIdentity = this.jobIdentity; + + if (workspaceLease) { + try { + workspaceRemoved = await cleanupSandboxWorkspace(workspaceLease); + } catch (error) { + workspaceRemoved = false; + this.log.error({ submissionDir: this.submissionDir, err: error }, 'Failed to clean up'); + } finally { + this.workspaceLease = undefined; + this.submissionDir = ''; + } + } + + if (jobIdentity) { + if (!workspaceLease || workspaceRemoved) { releaseJobIdentity(jobIdentity); - this.log.info( + } else { + retainWorkspaceCleanupUntilRemoved(workspaceLease, () => { + releaseJobIdentity(jobIdentity); + this.log.info( + { uid: jobIdentity.uid, gid: jobIdentity.gid, slot: jobIdentity.slot }, + 'Released retained sandbox job UID slot after workspace cleanup', + ); + }); + this.log.error( { uid: jobIdentity.uid, gid: jobIdentity.gid, slot: jobIdentity.slot }, - 'Released retained sandbox job UID slot after workspace cleanup', + 'Retaining sandbox job UID slot after failed workspace cleanup', ); - }); - this.log.error( - { uid: jobIdentity.uid, gid: jobIdentity.gid, slot: jobIdentity.slot }, - 'Retaining sandbox job UID slot after failed workspace cleanup', - ); + } + this.jobIdentity = undefined; } - this.jobIdentity = undefined; + outcome = workspaceRemoved ? 'removed' : 'retained'; + } finally { + recordCleanupResources({ + job: this.uuid, + mode: this.session ? 'session' : 'disposable', + outcome, + suppressSuccessLogs: this.isSynthetic, + }); } } } diff --git a/api/src/nsjail.ts b/api/src/nsjail.ts index 87a2b791..a9f02931 100644 --- a/api/src/nsjail.ts +++ b/api/src/nsjail.ts @@ -523,10 +523,10 @@ export async function execute(opts: ExecuteOptions, setupGate: NsJailSetupGate = }); const wallTime = Date.now() - startTime; - // Log memory metrics after each execution to track potential leaks. + // Pre-cleanup diagnostic only: Job.cleanup records post-cleanup resources. // Reads from /proc/self/cgroup to find the actual cgroup path, then reads // memory.current and memory.stat from that cgroup. - // Distinguishes real usage (anon) from reclaimable kernel page cache (file). + // `file` includes shmem/tmpfs and is not all reclaimable page cache. try { const cgroupLine = fs.readFileSync('/proc/self/cgroup', 'utf8').trim(); // cgroup v2 format: "0::" diff --git a/helm/codeapi/Chart.yaml b/helm/codeapi/Chart.yaml index 4e122bbf..dfcdd45b 100644 --- a/helm/codeapi/Chart.yaml +++ b/helm/codeapi/Chart.yaml @@ -3,7 +3,7 @@ apiVersion: v2 name: codeapi description: A Helm chart for Code Interpreter API - scalable code execution service type: application -version: 0.3.1 # Chart version (bump this when you change the chart) +version: 0.3.2 # Chart version (bump this when you change the chart) appVersion: "2.0.0" # App version (bump this when you change the app) # Keywords for searching diff --git a/helm/codeapi/README.md b/helm/codeapi/README.md index a063e597..f63950ae 100644 --- a/helm/codeapi/README.md +++ b/helm/codeapi/README.md @@ -55,6 +55,43 @@ platform rather than templated here: external ingress/service mesh, KEDA-style queue-depth autoscaling, and cloud-IAM secret delivery (the env hooks below cover all of them). +**Prometheus scraping.** With `metrics.enabled=true` and +`workerSandbox.enabled=true`, separate PodMonitors scrape the service-worker's `health` port and the sandbox-runner's +`sandbox` port at `/metrics`. Only the latter exports +`codeapi_sandbox_post_cleanup_*` metrics. Both inherit `metrics.interval` and +`metrics.scrapeTimeout`. Install the Prometheus Operator PodMonitor CRD and +configure Prometheus to discover this release's PodMonitors and namespace. + +The default NetworkPolicy allows runner ingress only from service-worker pods. +To permit scraping with `networkPolicy.enabled=true`, explicitly match your +trusted Prometheus namespace **and** pod labels in one peer, for example: + +```yaml +metrics: + enabled: true + sandboxRunner: + ingressFrom: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: observability + podSelector: + matchLabels: + app.kubernetes.io/name: prometheus +``` + +Replace the example labels with those of your scraper. The default empty list +adds no access, even when metrics are enabled. This rule is rendered only while +metrics and the worker/sandbox tier are enabled. It uses the configured runner +port and does not change sandbox egress. Any policies on the Prometheus side +must also allow the scrape connection. When chart network policies are disabled, +allow scraping through your platform's policies instead. + +NetworkPolicy cannot restrict access to an HTTP path. These peers can reach the +runner's execution endpoint as well, so do not use unrestricted peers such as +`{}` or disable execution-manifest verification to enable scraping. No extra +Service or externally exposed port is created. Chart upgrades do not modify +runner images or require changes to stored session state for this scrape wiring. + **Pairing-fence rollbacks.** Do not use a direct `helm rollback` from a chart revision containing the bridge pairing fence to an older revision. Helm runs rollback hooks from the target revision, so a pre-fence target cannot stop its diff --git a/helm/codeapi/templates/network-policy.yaml b/helm/codeapi/templates/network-policy.yaml index 2184e9f9..fe24123f 100644 --- a/helm/codeapi/templates/network-policy.yaml +++ b/helm/codeapi/templates/network-policy.yaml @@ -3,6 +3,7 @@ Network Policies for CodeAPI. The important boundary is sandbox-runner: - service-worker may call sandbox-runner. + - explicitly configured metrics peers may reach the same runner port. - sandbox-runner may call only egress-gateway (plus DNS if required). - egress-gateway delegates only to file-server and tool-call-server. */}} @@ -70,6 +71,15 @@ spec: ports: - protocol: TCP port: {{ .Values.workerSandbox.sandbox.port }} + {{- if and .Values.metrics.enabled .Values.workerSandbox.enabled }} + {{- with (.Values.metrics.sandboxRunner | default dict).ingressFrom }} + - from: + {{- toYaml . | nindent 6 }} + ports: + - protocol: TCP + port: {{ $.Values.workerSandbox.sandbox.port }} + {{- end }} + {{- end }} egress: - to: - namespaceSelector: {} diff --git a/helm/codeapi/templates/podmonitor.yaml b/helm/codeapi/templates/podmonitor.yaml index 405e3bc3..aee33f44 100644 --- a/helm/codeapi/templates/podmonitor.yaml +++ b/helm/codeapi/templates/podmonitor.yaml @@ -43,6 +43,28 @@ spec: matchLabels: {{- include "codeapi.serviceWorker.selectorLabels" . | nindent 6 }} {{- end }} +{{- if .Values.workerSandbox.enabled }} +--- +apiVersion: monitoring.coreos.com/v1 +kind: PodMonitor +metadata: + name: {{ include "codeapi.fullname" . }}-sandbox-runner + labels: + {{- include "codeapi.sandboxRunner.labels" . | nindent 4 }} +spec: + podMetricsEndpoints: + - path: /metrics + port: sandbox + {{- with .Values.metrics.interval }} + interval: {{ . }} + {{- end }} + {{- with .Values.metrics.scrapeTimeout }} + scrapeTimeout: {{ . }} + {{- end }} + selector: + matchLabels: + {{- include "codeapi.sandboxRunner.selectorLabels" . | nindent 6 }} +{{- end }} {{- if .Values.egressGateway.enabled }} --- apiVersion: monitoring.coreos.com/v1 diff --git a/helm/codeapi/values.yaml b/helm/codeapi/values.yaml index bafed305..0a242639 100644 --- a/helm/codeapi/values.yaml +++ b/helm/codeapi/values.yaml @@ -512,6 +512,11 @@ metrics: enabled: false # Set to true to create PodMonitors for Prometheus scraping interval: "30s" scrapeTimeout: "10s" + sandboxRunner: + # Explicit NetworkPolicy peers allowed to scrape the runner when metrics and + # networkPolicy are enabled. Empty preserves worker-only ingress. This port + # also serves /execute, so select only trusted Prometheus pods (see README). + ingressFrom: [] # ============================================================================= # NETWORK POLICIES diff --git a/tests/sandbox_runner_metrics.sh b/tests/sandbox_runner_metrics.sh new file mode 100755 index 00000000..df7c9ff3 --- /dev/null +++ b/tests/sandbox_runner_metrics.sh @@ -0,0 +1,147 @@ +#!/usr/bin/env bash +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +TMP_DIR="$(mktemp -d)" +trap 'rm -rf "$TMP_DIR"' EXIT +command -v helm >/dev/null + +# Like block_root_package_delivery.sh, render only this chart's templates so +# tests do not download Redis or MinIO dependencies or contact a cluster. +mkdir "$TMP_DIR/chart" +cp "$ROOT/helm/codeapi/values.yaml" "$TMP_DIR/chart/values.yaml" +cp -R "$ROOT/helm/codeapi/templates" "$TMP_DIR/chart/templates" +awk '/^dependencies:/{exit} {print}' "$ROOT/helm/codeapi/Chart.yaml" > "$TMP_DIR/chart/Chart.yaml" + +render() { + helm template telemetry "$TMP_DIR/chart" \ + --set executionManifest.privateKey=test \ + --set executionManifest.publicKey=test "$@" +} + +resource() { + local file="$1" kind="$2" name="$3" + awk -v kind="$kind" -v name="$name" ' + function emit() { if (foundKind && foundName) printf "%s", document } + /^---$/ { emit(); document=""; foundKind=0; foundName=0; next } + { document=document $0 "\n" } + $0 == "kind: " kind { foundKind=1 } + $0 == " name: " name { foundName=1 } + END { emit() } + ' "$file" +} + +contains() { + if ! grep -Fq -- "$2" "$1"; then + echo "Missing '$2' in $(basename "$1")" >&2 + exit 1 + fi +} + +check_monitor() { + local rendered="$1" component="$2" port="$3" + local monitor="$TMP_DIR/$component-monitor.yaml" + resource "$rendered" PodMonitor "telemetry-codeapi-$component" > "$monitor" + contains "$monitor" ' - path: /metrics' + contains "$monitor" " port: $port" + contains "$monitor" ' interval: 17s' + contains "$monitor" ' scrapeTimeout: 7s' + # Exact selectors must match the corresponding Deployment's pod labels, + # including this release, and not accidentally select the service worker. + sed -n '/^ selector:/,$p' "$monitor" > "$TMP_DIR/selector.yaml" + contains "$TMP_DIR/selector.yaml" ' app.kubernetes.io/name: codeapi' + contains "$TMP_DIR/selector.yaml" ' app.kubernetes.io/instance: telemetry' + contains "$TMP_DIR/selector.yaml" " app.kubernetes.io/component: $component" + resource "$rendered" Deployment "telemetry-codeapi-$component" > "$TMP_DIR/deployment.yaml" + contains "$TMP_DIR/deployment.yaml" " app.kubernetes.io/component: $component" + contains "$TMP_DIR/deployment.yaml" " - name: $port" +} + +render > "$TMP_DIR/default.yaml" +if grep -q '^kind: PodMonitor$' "$TMP_DIR/default.yaml"; then + echo 'metrics.enabled=false must not create PodMonitors' >&2 + exit 1 +fi + +render --set metrics.enabled=true --set metrics.interval=17s --set metrics.scrapeTimeout=7s \ + --set workerSandbox.sandbox.port=2345 > "$TMP_DIR/enabled.yaml" +check_monitor "$TMP_DIR/enabled.yaml" sandbox-runner sandbox +check_monitor "$TMP_DIR/enabled.yaml" service-worker health +contains "$TMP_DIR/deployment.yaml" ' app.kubernetes.io/instance: telemetry' +resource "$TMP_DIR/enabled.yaml" Deployment telemetry-codeapi-sandbox-runner > "$TMP_DIR/runner.yaml" +contains "$TMP_DIR/runner.yaml" ' containerPort: 2345' + +# Enabling metrics must not silently weaken sandbox isolation. +resource "$TMP_DIR/default.yaml" NetworkPolicy telemetry-codeapi-sandbox-runner > "$TMP_DIR/default-policy.yaml" +resource "$TMP_DIR/enabled.yaml" NetworkPolicy telemetry-codeapi-sandbox-runner > "$TMP_DIR/enabled-policy.yaml" +# Use the same configured runner port when comparing policies. +sed 's/port: 2345/port: 2000/' "$TMP_DIR/enabled-policy.yaml" > "$TMP_DIR/normalized-policy.yaml" +cmp "$TMP_DIR/default-policy.yaml" "$TMP_DIR/normalized-policy.yaml" + +cat > "$TMP_DIR/scraper.yaml" <<'YAML' +metrics: + sandboxRunner: + ingressFrom: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: metrics-test + podSelector: + matchLabels: + app.kubernetes.io/name: prometheus-test +YAML +render --set metrics.enabled=true --set workerSandbox.sandbox.port=2345 \ + -f "$TMP_DIR/scraper.yaml" > "$TMP_DIR/scraper-render.yaml" +resource "$TMP_DIR/scraper-render.yaml" NetworkPolicy telemetry-codeapi-sandbox-runner > "$TMP_DIR/scraper-policy.yaml" +contains "$TMP_DIR/scraper-policy.yaml" 'kubernetes.io/metadata.name: metrics-test' +contains "$TMP_DIR/scraper-policy.yaml" 'app.kubernetes.io/name: prometheus-test' +contains "$TMP_DIR/scraper-policy.yaml" 'app.kubernetes.io/component: service-worker' +# The namespace and pod selectors must be ANDed in ONE peer, not two ORed peers. +contains "$TMP_DIR/scraper-policy.yaml" ' - namespaceSelector:' +contains "$TMP_DIR/scraper-policy.yaml" ' podSelector:' +[[ "$(grep -Fc ' port: 2345' "$TMP_DIR/scraper-policy.yaml")" == 2 ]] +# Scraping changes only ingress, never the sandbox egress boundary. +sed -n '/^ egress:/,$p' "$TMP_DIR/scraper-policy.yaml" > "$TMP_DIR/scraper-egress.yaml" +sed -n '/^ egress:/,$p' "$TMP_DIR/default-policy.yaml" > "$TMP_DIR/default-egress.yaml" +cmp "$TMP_DIR/default-egress.yaml" "$TMP_DIR/scraper-egress.yaml" + +render -f "$TMP_DIR/scraper.yaml" > "$TMP_DIR/disabled-scraper.yaml" +resource "$TMP_DIR/disabled-scraper.yaml" NetworkPolicy telemetry-codeapi-sandbox-runner > "$TMP_DIR/disabled-scraper-policy.yaml" +cmp "$TMP_DIR/default-policy.yaml" "$TMP_DIR/disabled-scraper-policy.yaml" + +render --set metrics.enabled=true --set workerSandbox.enabled=false \ + -f "$TMP_DIR/scraper.yaml" > "$TMP_DIR/no-workers.yaml" +[[ -z "$(resource "$TMP_DIR/no-workers.yaml" PodMonitor telemetry-codeapi-sandbox-runner)" ]] +[[ -z "$(resource "$TMP_DIR/no-workers.yaml" PodMonitor telemetry-codeapi-service-worker)" ]] +resource "$TMP_DIR/no-workers.yaml" NetworkPolicy telemetry-codeapi-sandbox-runner > "$TMP_DIR/no-workers-policy.yaml" +if grep -q prometheus-test "$TMP_DIR/no-workers-policy.yaml"; then + echo 'disabled workerSandbox must not enable scrape ingress' >&2 + exit 1 +fi + +# Older values may omit the new subtree; this must remain worker-only ingress. +render --set metrics.enabled=true --set metrics.sandboxRunner=null > "$TMP_DIR/old-values.yaml" +resource "$TMP_DIR/old-values.yaml" NetworkPolicy telemetry-codeapi-sandbox-runner > "$TMP_DIR/old-values-policy.yaml" +cmp "$TMP_DIR/default-policy.yaml" "$TMP_DIR/old-values-policy.yaml" + +render --set metrics.enabled=true --set api.enabled=false > "$TMP_DIR/no-api.yaml" +[[ -n "$(resource "$TMP_DIR/no-api.yaml" PodMonitor telemetry-codeapi-sandbox-runner)" ]] +[[ -z "$(resource "$TMP_DIR/no-api.yaml" PodMonitor telemetry-codeapi-api)" ]] + +render --set metrics.enabled=true --set metrics.interval= --set metrics.scrapeTimeout= > "$TMP_DIR/no-timing.yaml" +resource "$TMP_DIR/no-timing.yaml" PodMonitor telemetry-codeapi-sandbox-runner > "$TMP_DIR/no-timing-monitor.yaml" +if grep -Eq 'interval:|scrapeTimeout:' "$TMP_DIR/no-timing-monitor.yaml"; then + echo 'empty scrape timing overrides must defer to Prometheus defaults' >&2 + exit 1 +fi + +render --set metrics.enabled=true --set networkPolicy.enabled=false \ + -f "$TMP_DIR/scraper.yaml" > "$TMP_DIR/no-policy.yaml" +[[ -n "$(resource "$TMP_DIR/no-policy.yaml" PodMonitor telemetry-codeapi-sandbox-runner)" ]] +if grep -q '^kind: NetworkPolicy$' "$TMP_DIR/no-policy.yaml"; then + echo 'networkPolicy.enabled=false must not create NetworkPolicies' >&2 + exit 1 +fi + +helm lint "$TMP_DIR/chart" --set executionManifest.privateKey=test --set executionManifest.publicKey=test \ + --set metrics.enabled=true -f "$TMP_DIR/scraper.yaml" +echo 'Sandbox runner metrics chart regressions passed'