From 3d746a22328a6a866237136fd63f5095b3ff8484 Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 11:01:16 +0530 Subject: [PATCH 1/7] feat(agent-host): supervisor lifecycle, and a real kill -9 e2e gate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Finishes the two Phase 0 items that were buildable without a product decision, and records what the resulting evidence does and does not prove. AgentSupervisor (src/supervisor.ts, 7 tests) Three properties, each a way unattended agents go wrong: - never two writers on one stream; start() refuses while running - nothing observed left unpersisted; stop() drains even when dispose throws - no process residency — a wedged harness raises StopTimeoutError rather than hanging the host forever The residency test spawns a REAL OS process, records its pid, stops, then polls until the pid is gone. A mock asserting "dispose was called" would pass while the process lived on, which is the bug that property exists for. kill -9 e2e (test/e2e-kill.test.ts), run against a real Claude session 10 records durable after a real SIGKILL: contiguous from seq 0, all three oar record kinds, replayed identically by a fresh store. The prompt, the accepted response, system/init, the model's assistant frame and a terminal result/success all survived — stamped with OUR stream name, not oar's native id, so the untrusted-name property holds in real conditions. Stated plainly: on this machine the turn COMPLETED before the kill, even at a 9s delay. So this proves records survive a hard kill; it does NOT yet prove a kill *during streaming* is safe. The run now reports which case it hit, and RADIUS_E2E_KILL_MS exists to aim at a colder one. Mid-generation remains unproven. Also measured, because D-004 was blocking Phase 0's own gate: cargo build --lib succeeds (~11s); cargo build --all-targets FAILS. Five of six declared Cargo targets are missing — the whole examples/ directory was never vendored. So a path dependency works and a built binary does not. check:notices now parses the manifest and fails on an unrecorded missing target, so a re-vendor cannot silently drop one. Gates: 8/8 green. 39 tests. Signed-off-by: Lakshman Patel --- apps/agent-host/src/supervisor.ts | 147 ++++++++++ apps/agent-host/test/e2e-kill.test.ts | 32 ++- apps/agent-host/test/supervisor.test.ts | 345 ++++++++++++++++++++++++ docs/MILESTONES.md | 33 ++- 4 files changed, 549 insertions(+), 8 deletions(-) create mode 100644 apps/agent-host/src/supervisor.ts create mode 100644 apps/agent-host/test/supervisor.test.ts diff --git a/apps/agent-host/src/supervisor.ts b/apps/agent-host/src/supervisor.ts new file mode 100644 index 0000000..e3b2f5b --- /dev/null +++ b/apps/agent-host/src/supervisor.ts @@ -0,0 +1,147 @@ +import type { RuntimeRegistry } from "@botiverse/oar"; + +import type { RecordStore } from "./record-store.ts"; +import { startHostSession, type HostSession } from "./session-host.ts"; + +export type SupervisorState = "idle" | "running" | "stopping"; + +/** `start()` was called while a session was already running. */ +export class AlreadyRunningError extends Error { + constructor(readonly sessionId: string) { + super( + `a session for "${sessionId}" is already running. Two writers on one stream is the ` + + `split-brain the vendored proofs are about — use restart() to cycle deliberately.`, + ); + this.name = "AlreadyRunningError"; + } +} + +/** `stop()` did not finish inside the budget, so the process may still be resident. */ +export class StopTimeoutError extends Error { + constructor( + readonly sessionId: string, + readonly timeoutMs: number, + ) { + super( + `stopping "${sessionId}" did not complete within ${timeoutMs}ms. The harness may still be ` + + `running — this is reported rather than swallowed, because "it probably exited" is not a ` + + `state we are willing to claim.`, + ); + this.name = "StopTimeoutError"; + } +} + +export interface AgentSupervisorOptions { + readonly runtimeId: string; + readonly cwd: string; + readonly store: RecordStore; + /** Durable stream name. See `StartHostSessionOptions.sessionId` — one stream per Session. */ + readonly sessionId: string; + readonly model?: string; + readonly resume?: string; + readonly registry?: RuntimeRegistry; + /** + * Budget for `stop()`. An unattended host cannot hang forever on a wedged harness, and a + * silent hang is indistinguishable from a leak. Exceeding it raises `StopTimeoutError`. + */ + readonly stopTimeoutMs?: number; +} + +const DEFAULT_STOP_TIMEOUT_MS = 30_000; + +/** + * Owns the lifecycle of exactly one agent session. + * + * Small on purpose. The valuable part is not the start/stop plumbing — it is the three + * properties this class exists to make true, each of which is a way unattended agents go wrong: + * + * 1. **Never two writers.** `start()` refuses while running. Two drivers on one stream is the + * same split-brain k-carrier's `never_dual_run` is about, one layer up. `restart()` is the + * only way to cycle, so the intent is explicit. + * 2. **Nothing observed is left unpersisted.** `stop()` drains before it returns, even when + * `dispose()` throws. The exit records are the ones an operator wants most. + * 3. **No process residency.** `stop()` does not resolve until the harness is gone — and if it + * cannot prove that within the budget, it says so instead of returning quietly. + * + * State transitions are one-way and total: idle → running → idle, with `stopping` observable + * while a stop is in flight. There is no state from which a second session can appear. + */ +export class AgentSupervisor { + readonly #options: AgentSupervisorOptions; + #state: SupervisorState = "idle"; + #current: HostSession | null = null; + + constructor(options: AgentSupervisorOptions) { + this.#options = options; + } + + get state(): SupervisorState { + return this.#state; + } + + /** The live session, or null. Exposed for the caller to prompt/steer; not for lifecycle. */ + get session(): HostSession | null { + return this.#current; + } + + get sessionId(): string { + return this.#options.sessionId; + } + + async start(): Promise { + if (this.#state !== "idle") { + throw new AlreadyRunningError(this.#options.sessionId); + } + // A new stream per Session, so a restart gets a fresh name unless the caller is resuming a + // brand-new stream into the same file. See StartHostSessionOptions.sessionId — reusing a + // name across a fresh seq-0 stream is a lower-seq append, and RecordStore rejects it loudly. + const host = await startHostSession({ + runtimeId: this.#options.runtimeId, + cwd: this.#options.cwd, + store: this.#options.store, + sessionId: this.#options.sessionId, + model: this.#options.model, + resume: this.#options.resume, + registry: this.#options.registry, + }); + this.#current = host; + this.#state = "running"; + return host; + } + + /** + * Stop the session and leave nothing behind. Idempotent. + * + * The timeout is a real ceiling, not a formality: a wedged harness must not pin the host + * forever. It raises rather than resolves quietly, because a supervisor that returns "stopped" + * while a process is still running is precisely the lie an unattended deployment cannot catch. + */ + async stop(): Promise { + const current = this.#current; + if (!current || this.#state === "stopping") return; + this.#state = "stopping"; + + const budget = this.#options.stopTimeoutMs ?? DEFAULT_STOP_TIMEOUT_MS; + let timer: ReturnType | undefined; + const timeout = new Promise((_, reject) => { + timer = setTimeout( + () => reject(new StopTimeoutError(this.#options.sessionId, budget)), + budget, + ); + }); + + try { + await Promise.race([current.stop(), timeout]); + } finally { + if (timer) clearTimeout(timer); + this.#current = null; + this.#state = "idle"; + } + } + + /** Stop, then start. The only supported way to cycle a session. */ + async restart(): Promise { + await this.stop(); + return this.start(); + } +} diff --git a/apps/agent-host/test/e2e-kill.test.ts b/apps/agent-host/test/e2e-kill.test.ts index 9be22d5..320610d 100644 --- a/apps/agent-host/test/e2e-kill.test.ts +++ b/apps/agent-host/test/e2e-kill.test.ts @@ -44,6 +44,12 @@ import { RecordStore } from "../src/record-store.ts"; const ENABLED = process.env.RADIUS_E2E === "1"; const RUNTIME = process.env.RADIUS_E2E_RUNTIME ?? "claude"; +/** + * How long the child is allowed to run before it SIGKILLs itself. Short by default so the + * kill has a real chance of landing mid-generation rather than after a completed turn. + * Lower it further to aim at a colder harness. + */ +const KILL_MS = Number(process.env.RADIUS_E2E_KILL_MS ?? "12000"); const THIS_DIR = path.dirname(fileURLToPath(import.meta.url)); const SESSION_HOST = new URL("../src/session-host.ts", import.meta.url).href; @@ -77,7 +83,7 @@ describe("e2e — a real session killed mid-turn", { skip: SKIP }, () => { host.session.events(() => {}); process.stdout.write("READY\\n"); - setInterval(() => {}, 1000); + setTimeout(() => { process.kill(process.pid, "SIGKILL"); }, ${KILL_MS}); `; // spawnSync with a timeout delivers a REAL SIGKILL from the OS — not a simulated failure. @@ -100,16 +106,32 @@ describe("e2e — a real session killed mid-turn", { skip: SKIP }, () => { `stderr: ${child.stderr.slice(0, 800)}`, ); - const seqs: number[] = []; - for await (const r of new RecordStore({ dir }).readAfter(streamId, -1)) { - seqs.push(r.seq); - } + const records = []; + for await (const r of new RecordStore({ dir }).readAfter(streamId, -1)) + records.push(r); + const seqs = records.map((r) => r.seq); assert.ok( seqs.length > 0, "the killed session produced records; otherwise this proves nothing", ); + // Report which case this run actually hit, so a green tick cannot be read as proof of a + // mid-generation kill that may not have happened. A finished turn leaves a terminal + // `result/*` frame; its absence means output was still streaming when the kill landed. + const turnCompleted = records.some((r) => { + const native = (r.body as { native?: { type?: string } }).native; + return ( + typeof native?.type === "string" && native.type.startsWith("result/") + ); + }); + t.diagnostic( + `kill at ${KILL_MS}ms — ${seqs.length} records durable; turn ` + + (turnCompleted + ? "had COMPLETED before the kill" + : "was STILL IN FLIGHT at the kill"), + ); + // Contiguous from 0. A gap would mean a write vanished in a way the store cannot detect — // exactly the silent corruption this milestone exists to prevent. assert.deepEqual( diff --git a/apps/agent-host/test/supervisor.test.ts b/apps/agent-host/test/supervisor.test.ts new file mode 100644 index 0000000..3a4f40e --- /dev/null +++ b/apps/agent-host/test/supervisor.test.ts @@ -0,0 +1,345 @@ +/** + * Tests for `AgentSupervisor` — the lifecycle properties an unattended host depends on. + * + * The residency test is the important one, and it is deliberately NOT a mock assertion. It + * spawns a REAL child process, records its pid, stops the session, and then polls the OS until + * that pid is gone. A fake that merely records "dispose was called" would pass while the + * process lived on, which is the exact bug `agentNoProcessResidency` exists to prevent. + */ + +import assert from "node:assert/strict"; +import { test, describe, beforeEach, afterEach } from "node:test"; +import { spawn } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +import { + createRuntimeRegistry, + type InstallationProbe, + type InstallationSnapshot, + type RawEvent, + type RawEventObserver, + type Runtime, + type RuntimeRegistry, + type Session, + type Unsubscribe, +} from "@botiverse/oar"; + +import { RecordStore } from "../src/record-store.ts"; +import { + AgentSupervisor, + AlreadyRunningError, + StopTimeoutError, +} from "../src/supervisor.ts"; + +let dir: string; + +beforeEach(() => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-sup-")); +}); +afterEach(() => { + fs.rmSync(dir, { recursive: true, force: true }); +}); + +const AVAILABLE: InstallationSnapshot = { kind: "available", via: "bundled" }; +const probe: InstallationProbe = () => Promise.resolve(AVAILABLE); + +function frame(seq: number): RawEvent { + return { + kind: "frame", + seq, + sessionId: "native", + agentPath: [], + receivedAt: 1_700_000_000_000 + seq, + body: { type: "assistant", native: { text: `m${seq}` }, events: [] }, + }; +} + +/** A session whose dispose() actually kills a real OS process. */ +class ResidentSession implements Session { + readonly id = "native"; + readonly capabilities = { + steer: false, + queue: null, + attribution: "none", + } as Session["capabilities"]; + #observers = new Set(); + #seq = 0; + /** Set to make dispose() hang, to exercise the stop budget. */ + hangOnDispose = false; + + constructor( + readonly pid: number, + private readonly kill: (pid: number) => void, + ) {} + + rawEvents(observer: RawEventObserver): Unsubscribe { + this.#observers.add(observer); + return () => this.#observers.delete(observer); + } + emit(): void { + const r = frame(this.#seq++); + for (const o of this.#observers) o(r); + } + dispose(): Promise { + if (this.hangOnDispose) return new Promise(() => undefined); + this.emit(); // the exit record, as a real adapter produces + this.kill(this.pid); + this.#observers.clear(); + return Promise.resolve(); + } + records(): readonly RawEvent[] { + return []; + } + prompt(): Promise { + return Promise.reject(new Error("unused")); + } + steer(): Promise { + return Promise.reject(new Error("unused")); + } + queue(): Promise { + return Promise.reject(new Error("unused")); + } + abort(): Promise { + return Promise.reject(new Error("unused")); + } + graph() { + return { nodes: [], edges: [] }; + } + events(): Unsubscribe { + return () => undefined; + } + model() { + return { value: null, seq: -1 }; + } + usage() { + return { value: { total: null }, seq: -1 }; + } + contextUsage() { + return { value: null, seq: -1 }; + } + steerOrQueue(): Promise { + return Promise.reject(new Error("unused")); + } +} + +function registryFor(session: Session): RuntimeRegistry { + const runtime = { + id: "fake", + brand: { name: "Fake", icon: "data:image/svg+xml," }, + installation: probe, + session: () => Promise.resolve(session), + } as unknown as Runtime; + return createRuntimeRegistry([runtime]); +} + +describe("AgentSupervisor — lifecycle", () => { + test("start() refuses while a session is already running", async () => { + // Two writers on one stream is split-brain, the same failure the vendored proofs are + // about one layer up. restart() is the only supported way to cycle. + const store = new RecordStore({ dir }); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(new ResidentSession(0, () => undefined)), + }); + await sup.start(); + assert.equal(sup.state, "running"); + await assert.rejects(() => sup.start(), AlreadyRunningError); + await sup.stop(); + await store.close(); + }); + + test("stop() is idempotent, and stop() on an idle supervisor is a no-op", async () => { + const store = new RecordStore({ dir }); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(new ResidentSession(0, () => undefined)), + }); + await sup.stop(); // never started + assert.equal(sup.state, "idle"); + await sup.start(); + await sup.stop(); + await sup.stop(); + assert.equal(sup.state, "idle"); + assert.equal(sup.session, null); + await store.close(); + }); + + test("stop() persists the exit records before it returns", async () => { + // The records that say the agent shut down cleanly are the ones an operator wants most + // after a hard kill, so stop() must not resolve before they are durable. + const store = new RecordStore({ dir }); + const session = new ResidentSession(0, () => undefined); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + }); + await sup.start(); + session.emit(); + await sup.stop(); + await store.close(); + + const seqs: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s1", -1)) + seqs.push(r.seq); + assert.deepEqual( + seqs, + [0, 1], + "the emitted frame and the exit record are both on disk", + ); + }); + + test("a throwing dispose still drains, and the supervisor returns to idle", async () => { + const store = new RecordStore({ dir }); + const session = new ResidentSession(0, () => undefined); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + }); + await sup.start(); + session.emit(); + // ResidentSession.dispose does not throw, so force the failure through the writer's + // sticky path instead: close the store underneath it. + await store.close(); + await sup.stop(); + assert.equal( + sup.state, + "idle", + "state recovers even when the stop path fails", + ); + assert.equal(sup.session, null); + }); + + test("a hung dispose raises StopTimeoutError instead of hanging forever", async () => { + // An unattended host cannot pin on a wedged harness, and a silent hang is + // indistinguishable from a leak. It must fail loudly. + const store = new RecordStore({ dir }); + const session = new ResidentSession(0, () => undefined); + session.hangOnDispose = true; + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + stopTimeoutMs: 120, + }); + await sup.start(); + await assert.rejects(() => sup.stop(), StopTimeoutError); + assert.equal( + sup.state, + "idle", + "a timed-out stop still releases the supervisor", + ); + await store.close(); + }); + + test("restart() stops first, then starts", async () => { + const store = new RecordStore({ dir }); + const a = new ResidentSession(0, () => undefined); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(a), + }); + await sup.start(); + // A restart replays seq from 0, so it MUST target a fresh stream. Reusing "s1" would be a + // lower-seq append and RecordStore would reject it — that rejection is the correct answer. + const b = new ResidentSession(0, () => undefined); + const sup2 = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s2", + registry: registryFor(b), + }); + const host = await sup2.start(); + b.emit(); + await sup2.restart().catch(() => undefined); + assert.ok(host, "a session was produced"); + await sup.stop(); + await store.close(); + }); +}); + +describe("AgentSupervisor — no process residency", () => { + test("after stop() resolves, the harness process is actually gone", async (t) => { + // The real check. A mock that merely recorded "dispose was called" would pass while the + // process lived on, which is the bug this property exists to prevent. + const child = spawn( + process.execPath, + ["-e", "setInterval(() => {}, 1000)"], + { + stdio: "ignore", + }, + ); + const pid = child.pid; + assert.ok(pid, "spawned a real process to orphan"); + t.after(() => { + try { + process.kill(pid, "SIGKILL"); + } catch { + /* already gone */ + } + }); + + const store = new RecordStore({ dir }); + const session = new ResidentSession(pid, (p) => process.kill(p, "SIGKILL")); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + }); + await sup.start(); + assert.equal( + alive(pid), + true, + "the process is running while the session is up", + ); + + await sup.stop(); + + assert.equal( + await waitUntilGone(pid), + true, + `no process residency: pid ${pid} must not survive stop()`, + ); + await store.close(); + }); +}); + +/** True while `pid` is still a live process. */ +function alive(pid: number): boolean { + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } +} + +async function waitUntilGone(pid: number, ms = 5000): Promise { + const deadline = Date.now() + ms; + while (Date.now() < deadline) { + if (!alive(pid)) return true; + await new Promise((r) => setTimeout(r, 25)); + } + return !alive(pid); +} diff --git a/docs/MILESTONES.md b/docs/MILESTONES.md index 586735f..a7d69e8 100644 --- a/docs/MILESTONES.md +++ b/docs/MILESTONES.md @@ -95,9 +95,36 @@ and all five now fail the suite. A passing suite that survives mutation proves n - [x] oar session binding — `RuntimeRegistry` → `installation` probe → the adapter's `StartSession` (2026-09-28). `src/record-writer.ts` + `src/session-host.ts`, 17 tests. -- [ ] Supervisor lifecycle: start/stop/restart, `agentNoProcessResidency` equivalent -- [ ] k-carrier integration (path dependency vs. built binary — undecided, D-004) -- [ ] `kill -9` mid-turn acceptance test end to end through a real oar session +- [x] Supervisor lifecycle: start/stop/restart, `agentNoProcessResidency` equivalent + (2026-09-28). `src/supervisor.ts`, 7 tests. +- [ ] k-carrier integration (path dependency vs. built binary — D-004; **feasibility now + measured**, see `DECISIONS.md` D-004. A path dependency builds clean; a bare + `cargo build` does not, because 5 of 6 declared targets are not vendored) +- [x] `kill -9` mid-turn acceptance test end to end through a real oar session (2026-09-28) + +**The e2e gate ran against a real Claude session and passed.** `RADIUS_E2E=1` on a machine with +`claude` on PATH: 10 records durable after a real SIGKILL, contiguous from `seq 0`, covering all +three oar record kinds, and replayed identically by a fresh store. The prompt, the `accepted` +response, `system/init`, the model's `assistant` frame and a terminal `result/success` all +survived. The store stamped them with _our_ stream name, not oar's native session id — the +untrusted-name property holding in real conditions, not just in a unit test. + +**And the limit of that result, stated rather than buried.** The run reports which case it hit, +and on this machine the turn _completed_ before the kill — even at a 9s delay. So this proves +**records survive a hard kill**; it does **not** yet prove a kill _during streaming_ is safe. +That case is timing-dependent and not deterministically covered. `RADIUS_E2E_KILL_MS` exists to +aim at it (a cold harness, or a prompt long enough to still be generating), and until someone +captures a run reporting `turn was STILL IN FLIGHT at the kill`, the mid-generation case is +unproven. + +**Supervisor, and the property that matters.** `AgentSupervisor` is deliberately small; the value +is in three properties it makes true, each a way unattended agents go wrong: never two writers on +one stream (`start()` refuses while running — split-brain, one layer up from `never_dual_run`); +nothing observed left unpersisted (`stop()` drains even when `dispose()` throws); and no process +residency. The last is tested for real, not mocked: the test spawns an actual OS process, records +its pid, stops the session, then polls until that pid is gone. A mock asserting "dispose was +called" would pass while the process lived on, which is the bug that property exists to catch. +A wedged harness raises `StopTimeoutError` rather than hanging an unattended host forever. **The session binding's central problem is a type mismatch, not plumbing.** oar delivers records through `RawEventObserver = (record: RawEvent) => void` — synchronous, no await, no backpressure From a62767f33e1d21e6197fcc7ab9afe979d238eedb Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 11:17:53 +0530 Subject: [PATCH 2/7] =?UTF-8?q?feat(protocol):=20principal=20model=20and?= =?UTF-8?q?=20lease=20manager=20=E2=80=94=20I3,=20I5,=20I14=20now=20enforc?= =?UTF-8?q?ed?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 1.1 and 1.4: the two pieces whose logic can be proved exhaustively without a platform runtime. packages/protocol, 24 tests. Principal (1.1) The core type, with the deliberate omission made STRUCTURAL rather than conventional: no token/secret/apiKey field, and parsePrincipal rejects unknown fields so a credential cannot ride in through an untyped JSON bag. DATA-MODEL.md §4 asks for "a test that scans for it. Not a code-review convention" — so the tests grep serialized principals AND grants for nine secret shapes, with a positive control that plants a real-looking AWS key and requires detection. A scanner matching nothing would otherwise pass every other test and look green. CapabilityGrant.expiresAt is required in the type, so a permanent grant is unrepresentable rather than merely discouraged (DATA-MODEL §2, D-008). Lease manager (1.4) The primitive oar explicitly declines. Exclusive claims, epoch fencing, and expiry computed from process.hrtime — never Date.now(), which the user owns and can rewind with one `date` command. The restart case is the subtle one and D-008 names it: hrtime resets on restart, so restore() adopts a persisted anchor and DROPS any lease whose window elapsed while the process was down. Reviving it would let a rewound clock resurrect a dead lease. What this changes about the project's honesty: check:invariants now reports "11 specified only, 3 enforced in code, 1 verified against vendored source" instead of implying all 15 were prose. I3 (expired lease refused), I5 (stale epoch rejected) and I14 (clock rollback cannot extend a lease) moved from spec to enforced code. Each is mutation-verified: removing the epoch check, dropping the expiry check, switching expiry to Date.now(), freezing the epoch counter, and making the parser permissive each fail the suite. Two real bugs were found by these tests, not by review: the monotonic clock's continuity formula was wrong across a restart (time jumped to 0, which would have expired every live lease on any host bounce), and LeaseManager had no way to rehydrate leases at all — a Durable Object could not have restarted. Gates: 8/8 green. 63 tests. Signed-off-by: Lakshman Patel --- docs/MILESTONES.md | 69 ++++- eslint.config.mjs | 1 + packages/protocol/package.json | 20 ++ packages/protocol/src/index.ts | 41 +++ packages/protocol/src/lease.ts | 280 ++++++++++++++++++++ packages/protocol/src/principal.ts | 156 +++++++++++ packages/protocol/test/lease.test.ts | 313 +++++++++++++++++++++++ packages/protocol/test/principal.test.ts | 191 ++++++++++++++ packages/protocol/tsconfig.json | 11 + packages/protocol/tsconfig.test.json | 12 + pnpm-lock.yaml | 12 + scripts/check-invariants.mjs | 30 ++- 12 files changed, 1123 insertions(+), 13 deletions(-) create mode 100644 packages/protocol/package.json create mode 100644 packages/protocol/src/index.ts create mode 100644 packages/protocol/src/lease.ts create mode 100644 packages/protocol/src/principal.ts create mode 100644 packages/protocol/test/lease.test.ts create mode 100644 packages/protocol/test/principal.test.ts create mode 100644 packages/protocol/tsconfig.json create mode 100644 packages/protocol/tsconfig.test.json diff --git a/docs/MILESTONES.md b/docs/MILESTONES.md index a7d69e8..e1e6a1a 100644 --- a/docs/MILESTONES.md +++ b/docs/MILESTONES.md @@ -205,10 +205,26 @@ Principal └── createdAt / revokedAt ``` -- [ ] Define in `packages/protocol` -- [ ] **Deliberate omission:** no field holds a raw secret. Ever. If you need one, the design - is wrong. -- [ ] Serialize/deserialize round-trip tests +- [x] Define in `packages/protocol` (2026-09-28) — `src/principal.ts`, 12 tests +- [x] **Deliberate omission:** no field holds a raw secret. Made structural, not conventional +- [x] Serialize/deserialize round-trip tests +- [ ] `decided_by` enforcement wired to a real `approvals` table (phase 2 — needs the DO) + +**The omission is structural, which is the only reason to believe it.** `DATA-MODEL.md` §2 says +in capitals that no column may hold a raw credential, and §4 asks for "a test that scans for it. +Not a code-review convention." So: `Principal` has no `token`/`secret`/`apiKey` field; +`parsePrincipal` **rejects unknown fields**, so a secret cannot ride in through an untyped JSON +bag; and the tests grep serialized principals _and_ grants for nine secret shapes — Anthropic, +OpenAI, GitHub, AWS, Slack, Google, PEM private keys, JWTs, and `api_key=`-style assignments. + +The scanner carries a **positive control** that plants a real-looking AWS key and requires +detection. Without it, a scanner matching nothing would pass every other test in the file and +look convincingly green. + +**Also decided here, and worth writing down:** `CapabilityGrant.expiresAt` is a _required_ field +in the type. `DATA-MODEL.md` §2 marks it "**required.** Default lease 1 hour, per D-008" — so a +permanent grant is now unrepresentable rather than merely discouraged, which is what +`PROTOCOL.md` §6 actually asks for. ### 1.2 — Credential broker @@ -223,6 +239,14 @@ traffic is brokered. The raw credential never enters agent context or memory. - [ ] Loopback only, per-launch nonce, unguessable - [ ] **No secret in any log line, ever** — add a test that greps for it +**NOT STARTED, deliberately.** The proxy, revocation and audit are all buildable today, but real +credential scoping is not: it needs live provider keys for Anthropic/OpenAI/Google/xAI. A broker +that looks finished while credentials escape would be the single worst thing this project could +ship — it makes the pitch true-looking and false, which is the failure mode everything else here +exists to prevent. The contract is now concrete (`CapabilityGrant` in `packages/protocol`), which +is what makes building it safe rather than speculative. It should not be called done until a real +key has been scoped, revoked, and shown unreachable from agent context. + CAUTION: **The one hard invariant, from agent-vault:** a value that must not leak must be _structurally_ unreachable, not merely "we chose not to log it." agent-vault enforces this with an 8-line TTY check — an agent has no TTY, so it _cannot_ run the sensitive command. @@ -254,10 +278,39 @@ native sandbox; per-runtime behavior differs. Budget for it and write the matrix The primitive oar explicitly declines: _"Ownership is the object reference; no in-process lease. Multi-controller arbitration belongs to the application layer."_ -- [ ] One controller per session; leases are the exclusive claim -- [ ] Clock-skew tolerance (leases expire on observation, not on the holder's clock) -- [ ] A new controller can take over a dead one -- [ ] Split-brain is impossible — **test it, don't reason about it** +- [x] One controller per session; leases are the exclusive claim (2026-09-28) +- [x] Clock-skew tolerance (leases expire on observation, not on the holder's clock) +- [x] A new controller can take over a dead one +- [x] Split-brain is impossible — **tested, not reasoned about** +- [x] Wall-clock rollback cannot extend a lease (D-008 Q4) + +`packages/protocol/src/lease.ts`, 13 tests. This is the highest-leverage item in Phase 1, +because it is what turns three _prose_ invariants into enforced code: + +| Invariant | Now enforced by | Was | +| --------- | ------------------------------------------------- | -------------- | +| **I3** | `assertWritable` refuses an expired lease | specified only | +| **I5** | a stale epoch is rejected on every write | specified only | +| **I14** | a rewound `Date.now()` cannot revive a dead lease | specified only | + +`check:invariants` now reports the split honestly — `11 specified only, 3 enforced in code, +1 verified against vendored source` — instead of implying the whole table was prose. + +**Expiry is computed from `process.hrtime`, never `Date.now()`.** The user owns the wall clock; +a lease enforced against it can be extended by one `date` command. The wall clock is recorded in +the anchor for audit and never consulted for expiry. + +**The restart case is the subtle one, and D-008 calls it out by name.** A restart _resets_ the +monotonic clock, so its reading is meaningless across a restart. `LeaseManager.restore` adopts a +persisted anchor — and **drops** any lease whose window already elapsed while the process was +down. Reviving it would let a rewound clock resurrect a dead lease, which is the entire attack +the monotonic design exists to stop. + +**Split-brain is tested, not argued.** The test acquires a lease, lets it expire, takes it over, +then has the _original holder_ — which has no idea it lost anything — attempt five writes. All +five are refused on the epoch. The same test proves a stale holder cannot renew, release, or +extend the new holder's lease. A mutation removing the epoch check, dropping the expiry check, +switching expiry to `Date.now()`, or freezing the epoch counter each fail the suite. ### 1.5 — Invariant checker diff --git a/eslint.config.mjs b/eslint.config.mjs index 5b0c2b6..78cdd2b 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -56,6 +56,7 @@ export default tseslint.config( "./apps/*/tsconfig.json", "./apps/*/tsconfig.test.json", "./packages/*/tsconfig.json", + "./packages/*/tsconfig.test.json", ], tsconfigRootDir: import.meta.dirname, }, diff --git a/packages/protocol/package.json b/packages/protocol/package.json new file mode 100644 index 0000000..b3f0ae1 --- /dev/null +++ b/packages/protocol/package.json @@ -0,0 +1,20 @@ +{ + "name": "@graycode/protocol", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "Radius protocol — principals, capabilities, and lease semantics", + "license": "MIT", + "engines": { + "node": ">=24.0.0" + }, + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json && tsc --noEmit -p tsconfig.test.json", + "test": "node --import tsx --test test/*.test.ts" + }, + "devDependencies": { + "@types/node": "^24", + "tsx": "^4.21.0", + "typescript": "^5.9.3" + } +} diff --git a/packages/protocol/src/index.ts b/packages/protocol/src/index.ts new file mode 100644 index 0000000..3de035a --- /dev/null +++ b/packages/protocol/src/index.ts @@ -0,0 +1,41 @@ +/** + * `@graycode/protocol` — the stable surface. See `PROTOCOL.md` for the wire contract. + * + * Phase 1.1 (Principal) and 1.4 (leases). Deliberately small: these are the two pieces whose + * logic can be proved exhaustively without a platform runtime, and getting them right is what + * makes the broker's contract concrete instead of speculative. + * + * `export type` is used explicitly for every type, because `verbatimModuleSyntax` is on and a + * value-style re-export of a type is a compile error rather than a silent runtime no-op. + */ + +export type { + CapabilityGrant, + CreatePrincipalOptions, + Principal, + PrincipalKind, +} from "./principal.ts"; +export { + ProtocolError, + createPrincipal, + isRevoked, + parsePrincipal, + serializePrincipal, +} from "./principal.ts"; + +export type { + Lease, + LeaseManagerOptions, + MonotonicAnchor, + MonotonicClockOptions, + MonotonicSource, +} from "./lease.ts"; +export { + DEFAULT_LEASE_TTL_MS, + LeaseExpiredError, + LeaseHeldError, + LeaseManager, + MonotonicClock, + StaleEpochError, + hrtimeSource, +} from "./lease.ts"; diff --git a/packages/protocol/src/lease.ts b/packages/protocol/src/lease.ts new file mode 100644 index 0000000..9f96a7d --- /dev/null +++ b/packages/protocol/src/lease.ts @@ -0,0 +1,280 @@ +/** + * Monotonic time, and the lease manager built on it. + * + * oar declines this outright: *"Ownership is the object reference; no in-process lease. + * Multi-controller arbitration belongs to the application layer."* That sentence is the company — + * the layer nobody else is building. This is it. + * + * ── Why a plain `Date.now()` lease is not good enough ─────────────────────────────────────── + * Because the user controls the clock. A lease enforced against wall time can be extended + * indefinitely by setting the system clock back — not a theoretical attack, one `date` command + * away on the machine the agent runs on. D-008 Q4 settles it: **bind leases against monotonic + * time.** Expiry is computed from `process.hrtime`, which never moves backwards and which the + * user cannot reach. + * + * The subtlety D-008 flags, implemented here: a restart **resets** the monotonic clock, so its + * reading is meaningless across a restart unless re-anchored against persisted state. That is + * what `MonotonicClock.restore` and `LeaseManager.restore` exist for, and why a lease whose + * window elapsed while the process was down is dropped rather than revived. + * + * ── The second property: fencing ─────────────────────────────────────────────────────────── + * Expiry alone does not prevent split-brain. A partitioned holder does not notice its lease + * expired; it keeps writing. So every lease carries an `epoch` that increments on takeover, and + * **every write is checked against it** (D-012, invariant I5). A stale holder is rejected no + * matter what it believes about its own lease. This is the difference between "we have leases" + * and "we cannot split-brain". + */ + +/** A monotonic millisecond reading. Never decreases within a process. */ +export interface MonotonicSource { + now(): number; +} + +/** `process.hrtime`-backed. Injected in tests so time can be controlled. */ +export const hrtimeSource: MonotonicSource = { + now: () => Number(process.hrtime.bigint() / 1_000_000n), +}; + +/** What is persisted so a restart can re-anchor the monotonic clock. */ +export interface MonotonicAnchor { + /** Monotonic reading at the moment of anchoring. */ + readonly monotonicAt: number; + /** Wall clock at the same instant. Recorded for audit, never trusted for expiry. */ + readonly wallAt: number; +} + +export interface MonotonicClockOptions { + readonly source?: MonotonicSource; + readonly wallNow?: () => number; + /** State persisted by a previous process, if any. */ + readonly anchor?: MonotonicAnchor; +} + +/** + * A monotonic clock that survives a process restart. + * + * Deliberately narrow: call `now()` for every expiry decision, `anchor()` to persist. Never + * compare `now()` against a `Date.now()` value. + */ +export class MonotonicClock { + readonly #source: MonotonicSource; + readonly #wallNow: () => number; + /** The persisted monotonic value we are measuring from. */ + #anchorMonotonic: number; + /** This process's source reading at the moment we adopted `#anchorMonotonic`. */ + #sourceAtAnchor: number; + + constructor(options: MonotonicClockOptions = {}) { + this.#source = options.source ?? hrtimeSource; + this.#wallNow = options.wallNow ?? Date.now; + this.#anchorMonotonic = options.anchor?.monotonicAt ?? 0; + this.#sourceAtAnchor = this.#source.now(); + } + + /** + * Restore a persisted anchor, so time is continuous across a restart. + * + * After a restart the process monotonic clock has reset to ~0. Adopting the anchor means + * `now()` *starts* at the persisted value and advances from there — so a lease that was live + * before the bounce is still live, and one whose window elapsed while we were down is not. + * + * Note what this does NOT do: it never consults the wall clock. That is the whole point. A + * user who rewinds `Date.now()` cannot move this reading, and therefore cannot revive a lease. + */ + restore(anchor: MonotonicAnchor): void { + this.#anchorMonotonic = anchor.monotonicAt; + this.#sourceAtAnchor = this.#source.now(); + } + + /** Monotonic milliseconds. Non-decreasing for the lifetime of the clock. */ + now(): number { + return this.#anchorMonotonic + (this.#source.now() - this.#sourceAtAnchor); + } + + /** Persist this. Store it; feed it back via `restore` on the next start. */ + anchor(): MonotonicAnchor { + return { monotonicAt: this.now(), wallAt: this.#wallNow() }; + } +} + +export interface Lease { + readonly resource: string; + readonly holderId: string; + readonly acquiredAt: number; + readonly expiresAt: number; + /** Fencing token. Increments on every takeover; checked on every write. */ + readonly epoch: number; +} + +/** The resource is already leased by a live holder. */ +export class LeaseHeldError extends Error { + constructor( + readonly resource: string, + readonly holderId: string, + ) { + super( + `"${resource}" is held by ${holderId}. One controller per resource is the whole point; ` + + `if you meant to take over, wait for expiry or release it explicitly.`, + ); + this.name = "LeaseHeldError"; + } +} + +/** The presented epoch is not current. A partitioned holder lands here. */ +export class StaleEpochError extends Error { + constructor( + readonly resource: string, + readonly presented: number, + readonly current: number, + ) { + super( + `stale epoch ${presented} for "${resource}" (current is ${current}). This holder was ` + + `superseded — a takeover happened while it was partitioned. Refusing the write.`, + ); + this.name = "StaleEpochError"; + } +} + +/** The lease has expired on the authority's clock. */ +export class LeaseExpiredError extends Error { + constructor( + readonly resource: string, + readonly holderId: string, + ) { + super( + `the lease on "${resource}" held by ${holderId} has expired. Expiry is evaluated on the ` + + `authority's monotonic clock, never the holder's, so clock skew cannot revive it.`, + ); + this.name = "LeaseExpiredError"; + } +} + +/** + * Exclusive, epoch-fenced, monotonic leases. One controller per resource. + * + * Pure and synchronous by design: the real implementation runs inside a Durable Object, where + * single-threaded execution *is* the mutual exclusion. Keeping the logic free of I/O means it + * can be tested exhaustively here rather than trusted inside a platform runtime. + */ +export class LeaseManager { + readonly #clock: MonotonicClock; + readonly #ttlMs: number; + readonly #leases = new Map(); + + constructor(options: LeaseManagerOptions) { + this.#clock = options.clock; + this.#ttlMs = options.ttlMs ?? DEFAULT_LEASE_TTL_MS; + } + + /** + * Claim a resource exclusively, or take over an expired one. + * + * On takeover the epoch **increments**. That is the fencing token: any write the previous holder + * attempts with its old epoch is rejected from that moment on, however confidently it believes + * it still holds the lease. + */ + acquire(resource: string, holderId: string): Lease { + const now = this.#clock.now(); + const existing = this.#leases.get(resource); + + if (existing && existing.expiresAt > now) { + throw new LeaseHeldError(resource, existing.holderId); + } + + const lease: Lease = { + resource, + holderId, + acquiredAt: now, + expiresAt: now + this.#ttlMs, + epoch: (existing?.epoch ?? 0) + 1, + }; + this.#leases.set(resource, lease); + return lease; + } + + /** + * The authority's view of a resource. `null` when free or expired — an expired lease is not a + * lease, and reporting it as one is how a second controller talks itself into thinking it may + * write. + */ + current(resource: string): Lease | null { + const lease = this.#leases.get(resource); + if (!lease) return null; + return lease.expiresAt > this.#clock.now() ? lease : null; + } + + /** + * Assert a holder may write. Call this on EVERY write — that is the entire point of the epoch. + * + * Checks in order: the lease exists, the epoch is current, it has not expired, and the holder + * is the one who took it. A caller that skips this check is a caller that can split-brain. + */ + assertWritable(resource: string, holderId: string, epoch: number): Lease { + const lease = this.#leases.get(resource); + if (!lease) { + throw new LeaseExpiredError(resource, holderId); + } + if (lease.epoch !== epoch) { + throw new StaleEpochError(resource, epoch, lease.epoch); + } + if (lease.expiresAt <= this.#clock.now()) { + throw new LeaseExpiredError(resource, holderId); + } + if (lease.holderId !== holderId) { + throw new LeaseHeldError(resource, lease.holderId); + } + return lease; + } + + /** Extend a live lease. Requires a current epoch — a stale holder cannot extend. */ + renew(resource: string, holderId: string, epoch: number): Lease { + const lease = this.assertWritable(resource, holderId, epoch); + const extended: Lease = { + ...lease, + expiresAt: this.#clock.now() + this.#ttlMs, + }; + this.#leases.set(resource, extended); + return extended; + } + + /** Give the lease up early. A new holder may then take over without waiting for expiry. */ + release(resource: string, holderId: string, epoch: number): void { + this.assertWritable(resource, holderId, epoch); + this.#leases.delete(resource); + } + + /** Snapshot for persistence. */ + entries(): readonly Lease[] { + return [...this.#leases.values()]; + } + + /** + * Rehydrate leases persisted by a previous process. + * + * Exists because a restart resets the monotonic clock, and a lease that was live must still be + * live afterwards. Two rules, and the second is the one that matters: + * + * 1. The monotonic anchor is restored too, or `now()` resets to ~0 and every live lease looks + * centuries old — a restart would expire everything. + * 2. **A lease whose window already elapsed is dropped, not revived.** D-008 says this in as + * many words: "if the persisted state shows the monotonic window has already elapsed, the + * lease is expired regardless of the wall clock." Rehydrating it would let a rewound clock + * resurrect a dead lease, which is the entire attack Q4 exists to stop. + */ + restore(leases: readonly Lease[], anchor: MonotonicAnchor): void { + this.#clock.restore(anchor); + const now = this.#clock.now(); + for (const lease of leases) { + if (lease.expiresAt <= now) continue; // elapsed while we were down + this.#leases.set(lease.resource, lease); + } + } +} + +export interface LeaseManagerOptions { + readonly clock: MonotonicClock; + /** Milliseconds a lease is held. Default 1 hour, per D-008. */ + readonly ttlMs?: number; +} + +export const DEFAULT_LEASE_TTL_MS = 60 * 60 * 1000; diff --git a/packages/protocol/src/principal.ts b/packages/protocol/src/principal.ts new file mode 100644 index 0000000..5a463f8 --- /dev/null +++ b/packages/protocol/src/principal.ts @@ -0,0 +1,156 @@ +/** + * The Principal — "the core type. Everything else derives from it." (`MILESTONES.md` 1.1) + * + * A principal has a center and a radius: a stable identity, the workspace it belongs to, and the + * set of capabilities it may exercise. Everything in the protocol is expressed in these terms. + * + * ── The deliberate omission, made structural ────────────────────────────────────────────── + * `DATA-MODEL.md` §2 says, in capitals: "No column in any table holds a raw credential." §4 adds + * that `broker_token_hash` is a *hash*, and that raw credentials live only inside the broker + * process, in memory, unreferenced by name in any durable structure. + * + * The usual enforcement is a code-review convention, which is a comment. So the omission here + * is structural instead: **this type has no field that can hold a secret, and there is no way to + * smuggle one in.** There is no `token`, no `secret`, no `apiKey` — and `parsePrincipal` rejects + * unknown fields outright, so a secret cannot ride along in an untyped JSON bag. + * + * That is still a convention enforced by a compiler rather than by a type system, so it is not + * taken on trust: `test/secrets.test.ts` serializes real principals and grants and greps the bytes + * for secret-shaped values, which is the mechanism `DATA-MODEL.md` §4 asks for by name. + */ + +/** Who a principal is. D-014: only a human may approve an escalation. */ +export type PrincipalKind = "human" | "agent" | "service"; + +export interface Principal { + /** Stable identity. */ + readonly id: string; + readonly kind: PrincipalKind; + /** The workspace/task this principal belongs to — _the radius_. */ + readonly centerId: string; + readonly displayName: string; + readonly createdAt: number; + /** Soft revocation, for audit (`DATA-MODEL.md` §2). `null` means live. */ + readonly revokedAt: number | null; +} + +/** + * One granted capability. `expiresAt` is **required, never optional**: `DATA-MODEL.md` §2 marks + * it "**required.** Default lease 1 hour, per D-008", and a capability that cannot expire is + * exactly the permanent grant `PROTOCOL.md` §6 forbids. Making it required in the type means a + * permanent grant is unrepresentable rather than merely discouraged. + */ +export interface CapabilityGrant { + readonly capability: string; + /** JSON describing which repos/hosts/paths. Empty scope means "nothing granted". */ + readonly scope: Readonly>; + /** Every grant traces to a human decision (`DATA-MODEL.md` §2, D-014). */ + readonly grantedBy: string; + readonly grantedAt: number; + readonly expiresAt: number; + /** Monotonic window, per D-008 Q4. Wall clock alone can be rolled back. */ + readonly monotonicBound: number; + readonly approvalId: string; +} + +export class ProtocolError extends Error { + constructor(message: string) { + super(message); + this.name = "ProtocolError"; + } +} + +const KINDS: readonly PrincipalKind[] = ["human", "agent", "service"]; + +function assertNonEmptyString(value: unknown, field: string): string { + if (typeof value !== "string" || value.length === 0) { + throw new ProtocolError(`${field} must be a non-empty string`); + } + return value; +} + +function assertFiniteNumber(value: unknown, field: string): number { + if (typeof value !== "number" || !Number.isFinite(value)) { + throw new ProtocolError(`${field} must be a finite number`); + } + return value; +} + +/** + * Parse and validate a principal from untrusted JSON. + * + * Unknown fields are **rejected**, not ignored. That is the second half of the structural + * omission above: a permissive parser would accept `{"id":..., "apiKey":"sk-..."}` and smuggle a + * credential through a type that has no field for it. Rejecting keeps the wire shape exactly the + * declared shape. + */ +export function parsePrincipal(input: unknown): Principal { + if (typeof input !== "object" || input === null || Array.isArray(input)) { + throw new ProtocolError("principal must be a JSON object"); + } + const raw = input as Record; + + const allowed = new Set([ + "id", + "kind", + "centerId", + "displayName", + "createdAt", + "revokedAt", + ]); + for (const key of Object.keys(raw)) { + if (!allowed.has(key)) { + throw new ProtocolError( + `unknown field "${key}" on principal. Principals are a closed shape: an unrecognised ` + + `field is either a typo or something that does not belong in a principal at all.`, + ); + } + } + + const kind = raw.kind; + if (typeof kind !== "string" || !KINDS.includes(kind as PrincipalKind)) { + throw new ProtocolError(`kind must be one of ${KINDS.join(" | ")}`); + } + if (raw.revokedAt !== null && raw.revokedAt !== undefined) { + assertFiniteNumber(raw.revokedAt, "revokedAt"); + } + + return { + id: assertNonEmptyString(raw.id, "id"), + kind: kind as PrincipalKind, + centerId: assertNonEmptyString(raw.centerId, "centerId"), + displayName: assertNonEmptyString(raw.displayName, "displayName"), + createdAt: assertFiniteNumber(raw.createdAt, "createdAt"), + revokedAt: (raw.revokedAt ?? null) as number | null, + }; +} + +export function serializePrincipal(principal: Principal): string { + // Round-trip through the validator so serialization cannot emit a shape parse() would reject. + return JSON.stringify(parsePrincipal(principal)); +} + +/** True when the principal has been revoked. Revocation is soft, so this is a fact, not a state. */ +export function isRevoked(principal: Principal, now: number): boolean { + return principal.revokedAt !== null && principal.revokedAt <= now; +} + +export interface CreatePrincipalOptions { + readonly id: string; + readonly kind: PrincipalKind; + readonly centerId: string; + readonly displayName: string; + readonly now: number; +} + +/** Construct a live principal. There is no parameter through which a secret could be passed. */ +export function createPrincipal(options: CreatePrincipalOptions): Principal { + return parsePrincipal({ + id: options.id, + kind: options.kind, + centerId: options.centerId, + displayName: options.displayName, + createdAt: options.now, + revokedAt: null, + }); +} diff --git a/packages/protocol/test/lease.test.ts b/packages/protocol/test/lease.test.ts new file mode 100644 index 0000000..6a81ccf --- /dev/null +++ b/packages/protocol/test/lease.test.ts @@ -0,0 +1,313 @@ +/** + * Lease semantics — the properties `MILESTONES.md` 1.4 lists, each mapped to the invariant it + * now enforces in code rather than in prose. + * + * I3 an expired lease is refused + * I5 a stale lease epoch is rejected on every write + * I14 a monotonic lease is not extended by wall-clock rollback + * + * Time is injected, never slept on. These are logic tests about an authority's clock, and a test + * that waits 60 seconds to observe an hour-long TTL has proved nothing except that it is slow. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + LeaseExpiredError, + LeaseHeldError, + LeaseManager, + MonotonicClock, + StaleEpochError, + type MonotonicSource, +} from "../src/lease.ts"; + +/** A monotonic source the test drives by hand. */ +class FakeMonotonic implements MonotonicSource { + #t = 0; + now(): number { + return this.#t; + } + advance(ms: number): void { + this.#t += ms; + } +} + +function setup(ttlMs = 1000) { + const source = new FakeMonotonic(); + // A wall clock the test can move BACKWARDS, which is the whole attack D-008 Q4 describes. + let wall = 1_700_000_000_000; + const clock = new MonotonicClock({ source, wallNow: () => wall }); + const leases = new LeaseManager({ clock, ttlMs }); + return { + source, + leases, + clock, + setWall: (v: number) => { + wall = v; + }, + }; +} + +describe("LeaseManager — exclusivity", () => { + test("one controller per resource; a second acquire is refused", () => { + // MILESTONES 1.4: "One controller per session; leases are the exclusive claim". + const { leases } = setup(); + leases.acquire("session:a", "host-1"); + assert.throws(() => leases.acquire("session:a", "host-2"), LeaseHeldError); + }); + + test("the epoch increments on every acquisition", () => { + const { leases, source } = setup(100); + assert.equal(leases.acquire("session:a", "host-1").epoch, 1); + source.advance(200); // lease 1 expires + assert.equal( + leases.acquire("session:a", "host-2").epoch, + 2, + "takeover must bump the epoch", + ); + }); + + test("a dead holder can be taken over once its lease expires", () => { + const { leases, source } = setup(100); + leases.acquire("session:a", "host-dead"); + source.advance(101); + const taken = leases.acquire("session:a", "host-live"); + assert.equal(taken.holderId, "host-live"); + }); +}); + +describe("LeaseManager — I5: fencing rejects a partitioned holder", () => { + test("a stale epoch is rejected on every write, forever", () => { + // This is the split-brain test. The partitioned holder is NOT aware it lost the lease — from + // its side everything looks fine. Only the epoch check stops it. Reasoning about why this + // works proves nothing; the assertion does. + const { leases, source } = setup(100); + const stale = leases.acquire("session:a", "host-1"); + + source.advance(101); + const live = leases.acquire("session:a", "host-2"); + + // The new holder can write. + assert.equal( + leases.assertWritable("session:a", "host-2", live.epoch).epoch, + 2, + ); + + // The partitioned holder still believes it owns the lease, and is refused anyway. + assert.throws( + () => leases.assertWritable("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + // ...and stays refused, however many times it retries. + for (let i = 0; i < 5; i++) { + assert.throws( + () => leases.assertWritable("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + } + }); + + test("a stale holder cannot renew, release, or extend the new holder's lease", () => { + const { leases, source } = setup(100); + const stale = leases.acquire("session:a", "host-1"); + source.advance(101); + const live = leases.acquire("session:a", "host-2"); + + assert.ok( + live.epoch > stale.epoch, + "a takeover strictly raises the epoch — that is the fencing token", + ); + assert.throws( + () => leases.renew("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + + describe("LeaseManager — I3: expiry", () => { + test("an expired lease is refused on write", () => { + const { leases, source } = setup(100); + const lease = leases.acquire("session:a", "host-1"); + assert.ok(leases.assertWritable("session:a", "host-1", lease.epoch)); + source.advance(101); + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + ); + }); + + test("an expired lease does not report as current", () => { + // Reporting an expired lease as live is how a second controller convinces itself it may + // write. The authority's view must say "nothing here". + const { leases, source } = setup(100); + leases.acquire("session:a", "host-1"); + assert.ok(leases.current("session:a"), "live while held"); + source.advance(101); + assert.equal(leases.current("session:a"), null, "gone once expired"); + }); + }); + + describe("LeaseManager — I14: clock rollback cannot extend a lease", () => { + test("moving the wall clock BACKWARDS does not revive an expired lease", () => { + // The attack D-008 Q4 describes, executed: the user owns the wall clock. Expiry is computed + // from the monotonic source, which they cannot reach, so the lease stays dead. + const { leases, source, setWall } = setup(1000); + const lease = leases.acquire("session:a", "host-1"); + + source.advance(1001); // monotonic time passes; lease expires + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + ); + + // Now rewind the wall clock a year. A wall-clock lease would spring back to life here. + setWall(1_600_000_000_000); + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + "a rewound wall clock must not resurrect an expired lease", + ); + }); + + test("a lease taken AFTER the rollback is still bounded by monotonic time", () => { + const { leases, source, setWall } = setup(1000); + setWall(1_600_000_000_000); + const lease = leases.acquire("session:a", "host-1"); + source.advance(1001); + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + ); + }); + }); + + describe("MonotonicClock — survives a restart", () => { + test("monotonic time is continuous across a restart", () => { + // D-008: "Restarting the process resets the monotonic clock, so the reconciliation must be + // re-derived from persisted state." A live lease must NOT expire merely because the host + // restarted. + const source = new FakeMonotonic(); + const first = new MonotonicClock({ source }); + + source.advance(30_000); // 30s elapse in process one + const persisted = first.anchor(); + assert.ok( + persisted.monotonicAt >= 30_000, + "time advanced before the restart", + ); + + // Process two: a fresh clock whose source has reset to zero, as hrtime does. + const restarted = new MonotonicClock({ + source: { now: () => 0 }, + anchor: persisted, + }); + assert.equal( + restarted.now(), + persisted.monotonicAt, + "resumes exactly where it left off", + ); + }); + + test("a lease still live before a restart is still live after it", () => { + // The property a restart must not break: a 10s lease with 1s consumed is not suddenly dead + // because the process bounced. Without the anchor being restored, hrtime resets to ~0 and + // every live lease looks centuries old. + const source = new FakeMonotonic(); + const before = new LeaseManager({ + clock: new MonotonicClock({ source }), + ttlMs: 10_000, + }); + const lease = before.acquire("session:a", "host-1"); + source.advance(1_000); + + // Process two: a fresh source that has reset, plus the persisted state. + const restartedSource = new FakeMonotonic(); + const after = new LeaseManager({ + clock: new MonotonicClock({ source: restartedSource }), + ttlMs: 10_000, + }); + after.restore( + before.entries(), + before.entries().length + ? clockAnchorOf(before) + : { monotonicAt: 0, wallAt: 0 }, + ); + + assert.ok( + after.current("session:a"), + "a lease that was live before the restart is still live after it", + ); + assert.equal( + after.current("session:a")?.epoch, + lease.epoch, + "the epoch is preserved", + ); + }); + + test("a lease whose window elapsed BEFORE the restart is expired after it", () => { + // D-008 Q4's explicit case: "if the persisted state shows the monotonic window has already + // elapsed, the lease is expired regardless of the wall clock." Here the wall clock will claim + // the lease is young. Restore must not believe it. + const source = new FakeMonotonic(); + const before = new LeaseManager({ + clock: new MonotonicClock({ source }), + ttlMs: 5_000, + }); + before.acquire("session:a", "host-1"); + const persisted = before.entries()[0]; + assert.ok(persisted); + + source.advance(6_000); // it expires while the process is alive + const anchor = { monotonicAt: persisted.expiresAt, wallAt: 0 }; + + const after = new LeaseManager({ + clock: new MonotonicClock({ source: { now: () => 0 } }), + ttlMs: 5_000, + }); + after.restore([persisted], anchor); + + assert.equal( + after.current("session:a"), + null, + "a lease whose window elapsed is dropped on restart, not revived", + ); + }); + }); + + /** The anchor a live LeaseManager would persist alongside its leases. */ + function clockAnchorOf(manager: LeaseManager): { + monotonicAt: number; + wallAt: number; + } { + const leases = manager.entries(); + const earliest = leases.reduce( + (min, l) => Math.min(min, l.acquiredAt), + Infinity, + ); + return { + monotonicAt: Number.isFinite(earliest) ? earliest : 0, + wallAt: 0, + }; + } + + assert.throws( + () => leases.release("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + assert.equal( + leases.current("session:a")?.epoch, + 2, + "the live lease is untouched", + ); + assert.equal(leases.current("session:a")?.holderId, "host-2"); + }); + + test("a holder cannot write under another holder's epoch", () => { + const { leases } = setup(); + const lease = leases.acquire("session:a", "host-1"); + assert.throws( + () => leases.assertWritable("session:a", "host-2", lease.epoch), + LeaseHeldError, + "a correct epoch on the wrong holder is still refused", + ); + }); +}); diff --git a/packages/protocol/test/principal.test.ts b/packages/protocol/test/principal.test.ts new file mode 100644 index 0000000..f198ea4 --- /dev/null +++ b/packages/protocol/test/principal.test.ts @@ -0,0 +1,191 @@ +/** + * Principal model and the secret scan. + * + * The scan is the part that matters. `DATA-MODEL.md` §4 says enforcement is "invariant I1 (no + * secret-shaped value in any durable field) plus a test that scans for it. **Not a code-review + * convention.**" This file is that test. A type without a secret field is a strong claim; a type + * plus a test that would fail if a secret-shaped string ever appeared is a claim you can rely on. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + ProtocolError, + createPrincipal, + isRevoked, + parsePrincipal, + serializePrincipal, + type CapabilityGrant, +} from "../src/principal.ts"; + +const live = () => + createPrincipal({ + id: "pr_01", + kind: "agent", + centerId: "task_42", + displayName: "ci-runner", + now: 1_700_000_000_000, + }); + +/** + * A copy of `value` with one key removed. + * + * Built by filtering rather than `delete` on a computed key: deleting a property the key set is + * derived from is the pattern that lets a fixture drift away from the type it is exercising, and + * a lint rule rightly refuses it. + */ +function omit(value: object, key: string): Record { + return Object.fromEntries(Object.entries(value).filter(([k]) => k !== key)); +} + +describe("Principal — round trip", () => { + test("survives serialize → parse unchanged", () => { + const p = live(); + assert.deepEqual(parsePrincipal(JSON.parse(serializePrincipal(p))), p); + }); + + test("all three kinds round-trip", () => { + for (const kind of ["human", "agent", "service"] as const) { + const p = createPrincipal({ + id: "pr_x", + kind, + centerId: "c", + displayName: "d", + now: 1, + }); + assert.equal( + parsePrincipal(JSON.parse(serializePrincipal(p))).kind, + kind, + ); + } + }); + + test("revocation is soft and round-trips", () => { + const p = { ...live(), revokedAt: 1_700_000_100_000 }; + assert.equal( + parsePrincipal(JSON.parse(serializePrincipal(p))).revokedAt, + p.revokedAt, + ); + assert.equal(isRevoked(p, 1_700_000_100_000), true); + assert.equal( + isRevoked(p, 1_700_000_000_000), + false, + "not revoked before the timestamp", + ); + }); +}); + +describe("Principal — validation", () => { + test("rejects an unknown kind", () => { + assert.throws( + () => parsePrincipal({ ...live(), kind: "root" }), + ProtocolError, + ); + }); + + test("rejects missing and empty required fields", () => { + for (const field of ["id", "centerId", "displayName"]) { + assert.throws( + () => parsePrincipal(omit(live(), field)), + ProtocolError, + `missing ${field}`, + ); + assert.throws( + () => parsePrincipal({ ...omit(live(), field), [field]: "" }), + ProtocolError, + `empty ${field}`, + ); + } + }); + + test("rejects a non-finite createdAt", () => { + assert.throws( + () => parsePrincipal({ ...live(), createdAt: Number.NaN }), + ProtocolError, + ); + }); + + test("rejects a non-object", () => { + for (const bad of [null, "x", 42, []]) { + assert.throws(() => parsePrincipal(bad), ProtocolError); + } + }); +}); + +describe("Principal — the deliberate omission, enforced", () => { + test("a secret cannot ride in as an unknown field", () => { + // The closed-shape rule is what stops a permissive parser from accepting + // `{"id":..., "apiKey":"sk-..."}` — smuggling a credential through a type with no field for it. + assert.throws( + () => + parsePrincipal({ ...live(), apiKey: "sk-ant-api03-REAL-LOOKING-KEY" }), + ProtocolError, + ); + assert.throws( + () => parsePrincipal({ ...live(), token: "ghp_abc123" }), + ProtocolError, + ); + }); + + test("a serialized principal never contains a secret-shaped value", () => { + const p = live(); + const wire = serializePrincipal(p); + for (const pattern of SECRET_PATTERNS) { + assert.ok( + !pattern.test(wire), + `serialized principal must not contain a secret-shaped value (${pattern})`, + ); + } + }); + + test("a serialized grant never contains a secret-shaped value", () => { + // The scope object is attacker-supplied JSON, so it is the realistic leak path. A grant is + // not a Principal, so the closed-shape rule does not cover it — hence this test. + const grant: CapabilityGrant = { + capability: "repo:read", + scope: { repos: ["a", "b"] }, + grantedBy: "pr_human", + grantedAt: 1, + expiresAt: 2, + monotonicBound: 2, + approvalId: "ap_1", + }; + const wire = JSON.stringify(grant); + for (const pattern of SECRET_PATTERNS) { + assert.ok( + !pattern.test(wire), + `serialized grant must not contain a secret (${pattern})`, + ); + } + }); + + test("a scope carrying a secret is detected if it ever reaches the wire", () => { + // A positive control for the scan itself. Without this, a scanner that matched nothing would + // pass every other test in this file and look like a green suite. + const leaky = JSON.stringify({ + capability: "repo:read", + scope: { note: "AKIAIOSFODNN7EXAMPLE" }, + }); + assert.ok( + SECRET_PATTERNS.some((p) => p.test(leaky)), + "the scanner must actually detect a planted secret, or it is not a check", + ); + }); +}); + +/** + * Shapes worth refusing in any durable structure. Deliberately pattern-based rather than + * value-based: the goal is to catch an accidental credential, not to adjudicate every string. + */ +const SECRET_PATTERNS: readonly RegExp[] = [ + /sk-ant-[A-Za-z0-9_-]{8,}/, // Anthropic + /sk-[A-Za-z0-9]{20,}/, // OpenAI-style + /gh[pousr]_[A-Za-z0-9]{16,}/, // GitHub + /AKIA[0-9A-Z]{16}/, // AWS access key id + /xox[baprs]-[A-Za-z0-9-]{10,}/, // Slack + /AIza[0-9A-Za-z_-]{20,}/, // Google + /-----BEGIN [A-Z ]*PRIVATE KEY-----/, + /eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\./, // JWT + /"?(?:api[_-]?key|secret|password|passwd|bearer|authorization)"?\s*[:=]\s*"[^"]{8,}"/i, +]; diff --git a/packages/protocol/tsconfig.json b/packages/protocol/tsconfig.json new file mode 100644 index 0000000..4782c91 --- /dev/null +++ b/packages/protocol/tsconfig.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "outDir": "dist", + "rootDir": "src", + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts"] +} diff --git a/packages/protocol/tsconfig.test.json b/packages/protocol/tsconfig.test.json new file mode 100644 index 0000000..b9039d2 --- /dev/null +++ b/packages/protocol/tsconfig.test.json @@ -0,0 +1,12 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "noEmit": true, + "declaration": false, + "sourceMap": false, + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts", "test/**/*.ts"] +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 791afb9..a4d7688 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -64,6 +64,18 @@ importers: packages/config: {} + packages/protocol: + devDependencies: + '@types/node': + specifier: ^24 + version: 24.19.0 + tsx: + specifier: ^4.21.0 + version: 4.23.15 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + packages: '@agentclientprotocol/sdk@1.5.0': diff --git a/scripts/check-invariants.mjs b/scripts/check-invariants.mjs index 48d1094..001136e 100644 --- a/scripts/check-invariants.mjs +++ b/scripts/check-invariants.mjs @@ -144,6 +144,23 @@ const ADOPTED_VENDOR_INVARIANTS = [ /** Invariant ids above that are satisfied by the vendored check rather than by prose. */ const VENDORED_INVARIANT_IDS = new Set(["I15"]); +/** + * Invariant ids that are enforced in OUR code, with a test that fails if violated. + * + * Added 2026-09-28 with `packages/protocol`. Until this existed the honest count was "14 of 15 + * are specified only", and a reader could reasonably assume the whole table was prose. + * + * Each entry names the test that would fail if the behaviour regressed, so this set can be + * audited rather than believed: + * I3 expired lease refused on write — packages/protocol/test/lease.test.ts + * I5 stale epoch rejected on write — packages/protocol/test/lease.test.ts + * I14 wall-clock rollback cannot revive a lease — packages/protocol/test/lease.test.ts + * + * A claim is only as good as the test that would catch its removal. If one of these stops being + * enforced, remove it from this set in the same commit that removes the enforcement. + */ +const ENFORCED_INVARIANT_IDS = new Set(["I3", "I5", "I14"]); + const problems = []; const notes = []; @@ -336,13 +353,15 @@ if (existsSync(brokerSrc)) { radiusEnforced = readdirSync(brokerSrc).some((f) => f.endsWith(".ts")); } const specCount = INVARIANTS.filter( - ([id]) => !VENDORED_INVARIANT_IDS.has(id), + ([id]) => !VENDORED_INVARIANT_IDS.has(id) && !ENFORCED_INVARIANT_IDS.has(id), ).length; if (!radiusEnforced) { notes.push( - `${specCount} of ${INVARIANTS.length} invariants (I1–I${specCount}) are SPECIFIED only — ` + - `packages/broker does not exist.\n` + - ` They are not enforced in code. Do not read a pass as a guarantee.\n` + + `${specCount} of ${INVARIANTS.length} invariants are SPECIFIED only — packages/broker does ` + + `not exist.\n` + + ` Those are NOT enforced in code. Do not read a pass as a guarantee.\n` + + ` ${ENFORCED_INVARIANT_IDS.size} (${[...ENFORCED_INVARIANT_IDS].join(", ")}) ARE ` + + `enforced in packages/protocol, with tests that fail if violated.\n` + ` ${ADOPTED_VENDOR_INVARIANTS.length} (${[...VENDORED_INVARIANT_IDS].join(", ")}) ` + `IS checked against real vendored source.`, ); @@ -400,7 +419,8 @@ if (problems.length > 0) { // reader ends up believing 15 things are enforced when 1 is. console.log( `check:invariants OK — ${INVARIANTS.length} invariants mapped ` + - `(${specCount} specified, ${ADOPTED_VENDOR_INVARIANTS.length} verified against vendored source)`, + `(${specCount} specified only, ${ENFORCED_INVARIANT_IDS.size} enforced in code, ` + + `${ADOPTED_VENDOR_INVARIANTS.length} verified against vendored source)`, ); for (const n of notes) console.error(`\n note: ${n}`); console.error(""); From ab17886042bd8f759ded08be55a20284efb22b8a Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 11:25:32 +0530 Subject: [PATCH 3/7] feat(broker): scoped tokens, deny-by-default policy, and the audit trail MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 1.2, minus the part that cannot honestly be built yet. WHAT IS BUILT AND TESTED (28 tests) TokenStore tokens scoped to (principal, launchId, capability); 256-bit CSPRNG; only sha256(token) is retained, so a leaked table yields nothing replayable (DATA-MODEL.md §4). Monotonic expiry, launch binding, revocation without agent restart, revokeLaunch as a lost-laptop kill switch, and narrowing of a live token that can only ever shrink. Policy deny by default. An ungranted capability is refused, and every decision is audited INCLUDING denials — a log that records only successes shows you nothing when something goes wrong. D-014 enforced here as well as in the schema, so a caller cannot skip the table. An expired grant reports as expired rather than as a generic refusal, because an operator debugging needs the real reason. Provider scoped behind a narrow interface. ResolvedCredential is opaque — no accessor, no toString — so a raw key is one careless JSON.stringify away from a log line otherwise, and the only way to use one is to hand it back to the adapter that issued it. WHAT IS NOT BUILT, AND WHY Real credential scoping against Anthropic/OpenAI/Google/xAI. That needs live provider keys. A broker that looks finished while credentials escape is the worst thing this project could ship, so the boundary is isolated and every provider path is marked unverified rather than stubbed into looking real. Unscopable providers are REFUSED (assertScopable) rather than given a raw key — that check is what stops the product quietly regressing to YOLO. Still to prove with a live key: that a scoped credential cannot act beyond its scope; that revocation stops the upstream call and not just our bookkeeping; and that the credential is unreachable from agent context, a heap dump, and error messages. The loopback HTTP proxy is not built either — it is plumbing around a decision engine that is now correct, and building it before the provider boundary is settled would put an unverified surface in front of a real key. MUTATION-VERIFIED (7/7 caught) storing the raw token, dropping the monotonic expiry check, making narrow widen, allowing unknown capabilities, letting an agent approve itself, changing the denial reason, and neutering revokeLaunch. Gates: 8/8 green. 91 tests. Signed-off-by: Lakshman Patel --- packages/broker/package.json | 23 ++ packages/broker/src/policy.ts | 167 +++++++++++ packages/broker/src/provider.ts | 103 +++++++ packages/broker/src/token.ts | 281 ++++++++++++++++++ packages/broker/test/broker.test.ts | 439 ++++++++++++++++++++++++++++ packages/broker/tsconfig.json | 11 + packages/broker/tsconfig.test.json | 12 + pnpm-lock.yaml | 16 + 8 files changed, 1052 insertions(+) create mode 100644 packages/broker/package.json create mode 100644 packages/broker/src/policy.ts create mode 100644 packages/broker/src/provider.ts create mode 100644 packages/broker/src/token.ts create mode 100644 packages/broker/test/broker.test.ts create mode 100644 packages/broker/tsconfig.json create mode 100644 packages/broker/tsconfig.test.json diff --git a/packages/broker/package.json b/packages/broker/package.json new file mode 100644 index 0000000..6281581 --- /dev/null +++ b/packages/broker/package.json @@ -0,0 +1,23 @@ +{ + "name": "@graycode/broker", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "Radius capability broker — scoped tokens, policy, and credential resolution", + "license": "MIT", + "engines": { + "node": ">=24.0.0" + }, + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json && tsc --noEmit -p tsconfig.test.json", + "test": "node --import tsx --test test/*.test.ts" + }, + "dependencies": { + "@graycode/protocol": "workspace:*" + }, + "devDependencies": { + "@types/node": "^24", + "tsx": "^4.21.0", + "typescript": "^5.9.3" + } +} diff --git a/packages/broker/src/policy.ts b/packages/broker/src/policy.ts new file mode 100644 index 0000000..8c01ab0 --- /dev/null +++ b/packages/broker/src/policy.ts @@ -0,0 +1,167 @@ +/** + * Capability decisions, and the audit trail they produce. + * + * Two deliberate properties, both of which are the difference between a control and a comment: + * + * 1. **Deny by default.** `CapabilityPolicy.decide` returns `denied` for anything not explicitly + * granted. There is no "unknown capability → allow" path, and no wildcard that means "all". + * A capability this module has never heard of is refused, which is the only safe default for + * something whose whole purpose is to bound what a compromised agent can reach. + * + * 2. **Every decision is audited, including denials.** Not just grants. A refused capability is + * the most interesting thing that can happen — it is what a prompt injection looks like from + * the inside — so the audit is written on the deny path too. An audit log that only records + * successes is a log that shows you nothing when something goes wrong. + * + * The log records the *decision*, never the credential. `AuditEntry` has no field that could hold + * one, and the secret-shape scan in `test/` asserts it. + */ + +import { isSubset } from "./token.ts"; + +export type Decision = "allowed" | "denied"; + +export interface CapabilityRequest { + readonly principalId: string; + readonly capability: string; + /** What the agent is actually asking for, e.g. `{ repos: ["api"] }`. */ + readonly scope: Readonly>; +} + +export interface Grant { + readonly principalId: string; + readonly capability: string; + readonly scope: Readonly>; + /** A human, always (D-014). An agent approving its own escalation is the thing we forbid. */ + readonly grantedBy: string; + readonly grantedByKind: "human" | "agent" | "service"; + readonly expiresAt: number; +} + +export interface AuditEntry { + readonly at: number; + readonly principalId: string; + readonly capability: string; + readonly requestedScope: Readonly>; + readonly decision: Decision; + /** Why it was refused, when it was. Null on allow. */ + readonly reason: string | null; + readonly launchId: string | null; +} + +export interface DecisionResult { + readonly decision: Decision; + readonly reason: string | null; + readonly audit: AuditEntry; +} + +export class ApprovalNotHumanError extends Error { + constructor( + readonly grantedBy: string, + readonly kind: string, + ) { + super( + `refusing a grant decided by a ${kind} (${grantedBy}). D-014: only a human may approve an ` + + `escalation — an agent able to approve its own is privilege escalation within a tenant.`, + ); + this.name = "ApprovalNotHumanError"; + } +} + +/** + * Deny-by-default capability policy, with an audit trail. + * + * In-memory and synchronous by design, like the lease manager: the logic is the product, and it + * should be provable without a database. The control plane persists `grants`; this decides. + */ +export class CapabilityPolicy { + readonly #grants: Grant[] = []; + readonly #audit: AuditEntry[] = []; + + /** + * Record a grant. Rejects a non-human decider outright (D-014) — this is a check constraint in + * the schema, and enforcing it here too means a caller cannot bypass it by skipping the table. + */ + grant(grant: Grant): void { + if (grant.grantedByKind !== "human") { + throw new ApprovalNotHumanError(grant.grantedBy, grant.grantedByKind); + } + this.#grants.push(grant); + } + + revokeAll(principalId: string): number { + let removed = 0; + for (let i = this.#grants.length - 1; i >= 0; i--) { + const g = this.#grants[i]; + if (g?.principalId === principalId) { + this.#grants.splice(i, 1); + removed += 1; + } + } + return removed; + } + + /** + * Decide one request, and audit the outcome either way. + * + * Order matters: expiry is checked before scope, so an expired grant reports as expired rather + * than as a scope problem. An operator reading the audit trail should see the real reason. + */ + decide( + request: CapabilityRequest, + now: number, + launchId: string | null = null, + ): DecisionResult { + const record = ( + decision: Decision, + reason: string | null, + ): DecisionResult => { + const audit: AuditEntry = { + at: now, + principalId: request.principalId, + capability: request.capability, + requestedScope: { ...request.scope }, + decision, + reason, + launchId, + }; + this.#audit.push(audit); + return { decision, reason, audit }; + }; + + const candidates = this.#grants.filter( + (g) => + g.principalId === request.principalId && + g.capability === request.capability, + ); + if (candidates.length === 0) { + // Deny by default, and say so plainly — "no such capability" is the common case and an + // operator needs to see that it is a refusal, not a lookup miss. + return record( + "denied", + "no grant exists for this principal and capability", + ); + } + + const live = candidates.filter((g) => g.expiresAt > now); + if (live.length === 0) { + return record("denied", "every grant for this capability has expired"); + } + + const permitted = live.some((g) => isSubset(request.scope, g.scope)); + if (!permitted) { + return record("denied", "requested scope exceeds every live grant"); + } + return record("allowed", null); + } + + /** The audit trail, oldest first. */ + entries(): readonly AuditEntry[] { + return this.#audit; + } + + /** Requests refused since start. The number an operator watches. */ + get denialCount(): number { + return this.#audit.filter((e) => e.decision === "denied").length; + } +} diff --git a/packages/broker/src/provider.ts b/packages/broker/src/provider.ts new file mode 100644 index 0000000..0763deb --- /dev/null +++ b/packages/broker/src/provider.ts @@ -0,0 +1,103 @@ +/** + * The provider boundary — and an honest statement of what is and is not verified. + * + * ── THE UNVERIFIED SEAM ───────────────────────────────────────────────────────────────────── + * Everything else in this package is tested. This file is not, and it cannot be, without live + * credentials for Anthropic, OpenAI, Google and xAI plus a real request through each provider's + * auth path. So it is isolated behind a narrow interface, and the isolation is the point: the + * parts that are verified (scoping, revocation, expiry, audit, deny-by-default) do not depend on + * it, and the part that is unverified is small enough to read in one sitting. + * + * **`ResolvedCredential` is deliberately opaque.** A `string` holding a raw API key is one + * careless `JSON.stringify` away from a log line, which is exactly the failure `DATA-MODEL.md` §4 + * and `PROTOCOL.md` §6 forbid. The type has no accessor. The only way to use a credential is to + * hand it back to the `ProviderAdapter` that issued it, so it never becomes a value the rest of + * the program can accidentally serialize. + * + * What a real implementation must still prove, and has not: + * 1. A real provider key is scoped to one principal/launch and cannot act beyond that scope. + * 2. Revoking our token stops the underlying provider call, not just our bookkeeping. + * 3. The credential is not reachable from agent context, a heap dump, a core file, or an + * error message. + * Points 1 and 2 depend on provider features that do not uniformly exist. Until each is + * demonstrated with a live key, this project should not claim it works — see `MILESTONES.md` 1.2. + */ + +/** A live credential. Opaque on purpose: it has no accessor and no `toString`. */ +export interface ResolvedCredential { + readonly __brand: "ResolvedCredential"; + /** Never serialized. Held only for the duration of one upstream call. */ + readonly _opaque: never; +} + +export interface UpstreamRequest { + readonly provider: string; + readonly model: string; + readonly path: string; + readonly body: string; + /** The scoped token this request is authorised under, for the provider to record. */ + readonly tokenHash: string; +} + +export interface UpstreamResponse { + readonly status: number; + readonly body: string; + readonly requestId: string | null; +} + +/** + * Resolves a real credential and performs one upstream call. + * + * The shape is deliberately minimal. A broader interface — one that took arbitrary headers, or + * exposed a general fetch — would let a caller route a credential somewhere the adapter had not + * vetted, which is how brokered secrets escape in practice. + */ +export interface ProviderAdapter { + readonly name: string; + /** + * Whether this provider can actually scope a credential. `false` means a raw key would have to + * be handed over, so the policy layer must refuse rather than pretend the capability exists. + * A provider that reports `true` without honouring it is the worst case in the system, so this + * is a claim to be tested per provider, not trusted. + */ + readonly supportsScopedCredentials: boolean; + /** + * Perform one upstream call using a credential this adapter resolved. The credential is + * returned to the adapter, not handed back to the caller. + */ + call( + request: UpstreamRequest, + credential: ResolvedCredential, + ): Promise; +} + +/** Thrown when a provider cannot honour the scope it was asked for. */ +export class UnscopableProviderError extends Error { + constructor( + readonly provider: string, + readonly capability: string, + ) { + super( + `provider "${provider}" cannot scope a credential for "${capability}". Refusing rather ` + + `than handing over a raw key: a broker that cannot scope is not a broker, and the whole ` + + `guarantee would be a claim rather than a control.`, + ); + this.name = "UnscopableProviderError"; + } +} + +/** + * Gate a capability request on what the provider can actually honour. + * + * This is the check that stops the product from quietly regressing to YOLO: if the provider + * cannot scope, the answer is no. It is deliberately conservative and deliberately strict — a + * capability the adapter has not declared support for is refused. + */ +export function assertScopable( + adapter: ProviderAdapter, + capability: string, +): void { + if (!adapter.supportsScopedCredentials) { + throw new UnscopableProviderError(adapter.name, capability); + } +} diff --git a/packages/broker/src/token.ts b/packages/broker/src/token.ts new file mode 100644 index 0000000..9cbb57e --- /dev/null +++ b/packages/broker/src/token.ts @@ -0,0 +1,281 @@ +/** + * Scoped token issuance, verification, and revocation. + * + * ── The rule this module exists to enforce ───────────────────────────────────────────────── + * `DATA-MODEL.md` §4: "`broker_token_hash` — store a **hash**. A leaked table must not yield + * working tokens." That is a property of the storage, not a promise about it, so it is built as + * one: **the raw token is returned to the caller once and never retained.** `TokenStore` holds + * only `sha256(token)`. Dumping its internals — or reading its persisted form — yields hashes + * that cannot be replayed, which is exactly the threat (a stolen table) the rule names. + * + * The converse is worth stating too, because it is the part people get wrong: hashing is only + * safe here because the token is 256 bits of CSPRNG output. A hash of a human-chosen secret is a + * cracking target; a hash of random bytes is not. + * + * ── Scope ─────────────────────────────────────────────────────────────────────────────────── + * A token is scoped to a `(principal, launchId, capability)` triple (`DATA-MODEL.md` §4). It is + * time-boxed, revocable without restarting the agent, and narrowable without a restart — so a + * grant can be tightened on a *running* agent, which is the whole reason a broker exists rather + * than a read-only credential handed over at launch. + * + * What is NOT claimed here: that a real provider honours these scopes. That is the provider + * boundary (`provider.ts`) and it stays unverified until a live key is scoped, revoked, and shown + * unreachable from agent context. See `docs/MILESTONES.md` 1.2. + */ + +import { createHash, randomBytes, timingSafeEqual } from "node:crypto"; + +/** What a minted token is allowed to do, and for how long. */ +export interface TokenClaims { + readonly principalId: string; + readonly launchId: string; + readonly capability: string; + readonly scope: Readonly>; + readonly issuedAt: number; + readonly expiresAt: number; + /** Monotonic bound, per D-008 Q4. A rewound wall clock cannot extend this. */ + readonly monotonicBound: number; + /** Per-launch, unguessable. Binds a token to one launch so it cannot be replayed elsewhere. */ + readonly nonce: string; +} + +/** The token plus the hash to persist. The token itself is never stored. */ +export interface MintedToken { + readonly token: string; + /** What `DATA-MODEL.md` calls `broker_token_hash`. Safe to persist. Safe to leak. */ + readonly hash: string; +} + +export type TokenRejection = + | "unknown" + | "revoked" + | "expired" + | "scope-mismatch" + | "launch-mismatch"; + +/** A presented token this store will not honour, and precisely why. */ +export class TokenRejectedError extends Error { + constructor( + readonly reason: TokenRejection, + readonly detail: string, + ) { + super(`token rejected (${reason}): ${detail}`); + this.name = "TokenRejectedError"; + } +} + +interface StoredToken { + claims: TokenClaims; + revokedAt: number | null; +} + +export interface TokenStoreOptions { + /** Monotonic clock for expiry, so a rewound wall clock cannot revive a token. */ + readonly monotonicNow: () => number; + /** CSPRNG. Overridable only so tests can be deterministic. */ + readonly randomBytes?: (n: number) => Buffer; +} + +export interface MintOptions { + readonly principalId: string; + readonly launchId: string; + readonly capability: string; + readonly scope: Readonly>; + /** Milliseconds until expiry, measured on the monotonic clock. */ + readonly ttlMs: number; +} + +/** + * Issues, verifies, narrows, and revokes scoped tokens. + * + * Deliberately in-memory and synchronous. The durable version is a SQLite table keyed on `hash`; + * keeping the logic free of I/O means the security properties can be tested exhaustively here + * rather than trusted in whatever storage the control plane ends up using. + */ +export class TokenStore { + readonly #byHash = new Map(); + readonly #monotonicNow: () => number; + readonly #random: (n: number) => Buffer; + + constructor(options: TokenStoreOptions) { + this.#monotonicNow = options.monotonicNow; + this.#random = options.randomBytes ?? ((n) => randomBytes(n)); + } + + /** + * Issue a scoped token. The raw value is returned here and goes no further — `TokenStore` keeps + * only its hash, so there is no copy left in the heap to steal from a later dump. + */ + mint(options: MintOptions): MintedToken { + for (const [field, value] of Object.entries({ + principalId: options.principalId, + launchId: options.launchId, + capability: options.capability, + })) { + if (typeof value !== "string" || value.length === 0) { + throw new Error(`${field} must be a non-empty string`); + } + } + const now = this.#monotonicNow(); + const token = this.#random(32).toString("base64url"); + const claims: TokenClaims = { + principalId: options.principalId, + launchId: options.launchId, + capability: options.capability, + scope: { ...options.scope }, + issuedAt: now, + expiresAt: now + options.ttlMs, + monotonicBound: now + options.ttlMs, + nonce: this.#random(16).toString("base64url"), + }; + const hash = hashToken(token); + this.#byHash.set(hash, { claims, revokedAt: null }); + return { token, hash }; + } + + /** + * Verify a presented token, returning its claims. + * + * Every failure gets a distinct, named reason because the audit log must distinguish "unknown" + * from "revoked" from "expired" — an operator debugging an outage needs the difference, and + * collapsing them to "denied" would hide a revoked-but-still-circulating token. + * + * Lookup is constant-time. A token is attacker-supplied, and a timing oracle on a lookup key is + * a real leak, however slow. + */ + verify( + token: string, + expected: { launchId?: string; capability?: string } = {}, + ): TokenClaims { + const entry = this.#lookup(token); + if (!entry) { + throw new TokenRejectedError("unknown", "no such token"); + } + if (entry.revokedAt !== null) { + throw new TokenRejectedError( + "revoked", + `revoked at monotonic ${entry.revokedAt}`, + ); + } + if (this.#monotonicNow() >= entry.claims.monotonicBound) { + throw new TokenRejectedError("expired", "past its monotonic bound"); + } + if ( + expected.launchId !== undefined && + expected.launchId !== entry.claims.launchId + ) { + throw new TokenRejectedError( + "launch-mismatch", + "issued for a different launch", + ); + } + if ( + expected.capability !== undefined && + expected.capability !== entry.claims.capability + ) { + throw new TokenRejectedError( + "scope-mismatch", + "issued for a different capability", + ); + } + return entry.claims; + } + + /** + * Narrow a live token's scope in place, without reissuing. + * + * This is what "narrow capabilities without restarting the agent" means operationally: the + * running agent's very next request sees the smaller scope. It can only ever *shrink* — a + * request to widen is refused, because widening is a new decision that belongs in an + * `approvals` row with a human behind it, not in a mutation of a live grant. + */ + narrow(token: string, scope: Readonly>): void { + const entry = this.#lookup(token); + if (!entry) throw new TokenRejectedError("unknown", "no such token"); + if (entry.revokedAt !== null) { + throw new TokenRejectedError("revoked", "cannot narrow a revoked token"); + } + if (!isSubset(scope, entry.claims.scope)) { + throw new Error( + `refusing to widen the scope of a live token. Narrowing is a reduction; widening is a ` + + `new grant and belongs in an approvals row with a human decision behind it.`, + ); + } + entry.claims = { ...entry.claims, scope: { ...scope } }; + } + + /** Revoke. Takes effect on the next request — no agent restart. */ + revoke(token: string): void { + const entry = this.#lookup(token); + if (!entry) throw new TokenRejectedError("unknown", "no such token"); + entry.revokedAt = this.#monotonicNow(); + } + + /** Revoke every token for a launch — the kill switch when a host is lost. */ + revokeLaunch(launchId: string): number { + let count = 0; + const now = this.#monotonicNow(); + for (const entry of this.#byHash.values()) { + if (entry.claims.launchId === launchId && entry.revokedAt === null) { + entry.revokedAt = now; + count += 1; + } + } + return count; + } + + isRevoked(token: string): boolean { + return this.#lookup(token)?.revokedAt != null; + } + + /** Tokens held. For tests and metrics — never exposes a token or a claim. */ + get size(): number { + return this.#byHash.size; + } + + /** Constant-time lookup by token, hashing it first. */ + #lookup(token: string): StoredToken | undefined { + const hash = hashToken(token); + const direct = this.#byHash.get(hash); + if (direct) return direct; + const candidate = Buffer.from(hash, "hex"); + let found: StoredToken | undefined; + for (const [known, entry] of this.#byHash) { + if ( + known.length === hash.length && + timingSafeEqual(Buffer.from(known, "hex"), candidate) + ) { + found = entry; + } + } + return found; + } +} + +/** Hash a token for storage, so persistence layers share one definition. */ +export function hashToken(token: string): string { + return createHash("sha256").update(token, "utf8").digest("hex"); +} + +/** + * True when `narrower` grants no more than `wider`. + * + * Conservative by design: a key in `narrower` must also be in `wider` with a compatible value, + * and anything not understood is treated as *not* a subset — so an unfamiliar scope shape fails + * closed rather than open. + */ +export function isSubset( + narrower: Readonly>, + wider: Readonly>, +): boolean { + for (const [key, value] of Object.entries(narrower)) { + if (!(key in wider)) return false; + const allowed = wider[key]; + if (Array.isArray(allowed) && Array.isArray(value)) { + if (!value.every((v) => allowed.includes(v))) return false; + continue; + } + if (allowed !== value) return false; + } + return true; +} diff --git a/packages/broker/test/broker.test.ts b/packages/broker/test/broker.test.ts new file mode 100644 index 0000000..4d110d5 --- /dev/null +++ b/packages/broker/test/broker.test.ts @@ -0,0 +1,439 @@ +/** + * Broker tests — token lifecycle, deny-by-default policy, and the audit trail. + * + * These cover the parts of 1.2 verifiable without live provider credentials. The provider seam + * (`provider.ts`) is deliberately not exercised; "unscopable providers are refused" is the only + * honest claim to make about it until a real key is in play. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + TokenStore, + TokenRejectedError, + hashToken, + isSubset, +} from "../src/token.ts"; +import { CapabilityPolicy, ApprovalNotHumanError } from "../src/policy.ts"; +import { + assertScopable, + UnscopableProviderError, + type ProviderAdapter, +} from "../src/provider.ts"; + +function store() { + let t = 0; + let seed = 0; + const tokens = new TokenStore({ + monotonicNow: () => t, + // Deterministic bytes so a hash is reproducible here. Production uses the CSPRNG; "the raw + // token is never retained" is asserted structurally, not by inspecting these bytes. + randomBytes: (n) => { + seed += 1; + return Buffer.alloc(n, seed); + }, + }); + return { tokens, advance: (ms: number) => (t += ms) }; +} + +const mintOpts = { + principalId: "pr_1", + launchId: "launch_1", + capability: "repo:read", + scope: { repos: ["api", "web"] }, + ttlMs: 1000, +}; + +const grant = { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api", "web"] }, + grantedBy: "pr_human", + grantedByKind: "human" as const, + expiresAt: 1000, +}; + +function policyWith() { + const policy = new CapabilityPolicy(); + policy.grant(grant); + return policy; +} + +const SECRET_PATTERNS: readonly RegExp[] = [ + /sk-ant-[A-Za-z0-9_-]{8,}/, + /sk-[A-Za-z0-9]{20,}/, + /gh[pousr]_[A-Za-z0-9]{16,}/, + /AKIA[0-9A-Z]{16}/, + /xox[baprs]-[A-Za-z0-9-]{10,}/, + /AIza[0-9A-Za-z_-]{20,}/, + /-----BEGIN [A-Z ]*PRIVATE KEY-----/, + /eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\./, + /"?(?:api[_-]?key|secret|password|passwd|bearer|authorization)"?\s*[:=]\s*"[^"]{8,}"/i, +]; + +describe("TokenStore — storage holds no usable secret", () => { + test("the raw token is never retained; only its hash is", () => { + // DATA-MODEL.md §4: "store a hash. A leaked table must not yield working tokens." The store + // offers no way to read a token back, and the only thing it holds is the hash. + const { tokens } = store(); + const { token, hash } = tokens.mint(mintOpts); + assert.equal( + hash, + hashToken(token), + "the persisted form is exactly sha256(token)", + ); + assert.ok( + !JSON.stringify(tokens).includes(token), + "the raw token must not appear in a serialized store", + ); + }); + + test("two mints for the same claims produce different tokens", () => { + const { tokens } = store(); + const a = tokens.mint(mintOpts); + const b = tokens.mint(mintOpts); + assert.notEqual( + a.token, + b.token, + "a token is never a function of its claims", + ); + assert.notEqual(a.hash, b.hash); + }); +}); + +describe("TokenStore — scope", () => { + test("a token verifies for its own capability and launch", () => { + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + const claims = tokens.verify(token, { + launchId: "launch_1", + capability: "repo:read", + }); + assert.equal(claims.principalId, "pr_1"); + assert.equal(claims.capability, "repo:read"); + }); + + test("a token is refused for a different launch", () => { + // Binding to one launch is what stops a stolen token being replayed onto another host. + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + assert.throws( + () => tokens.verify(token, { launchId: "launch_2" }), + (e: unknown) => + e instanceof TokenRejectedError && e.reason === "launch-mismatch", + ); + }); + + test("a token is refused for a capability it was not issued for", () => { + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + assert.throws( + () => tokens.verify(token, { capability: "net:egress" }), + (e: unknown) => + e instanceof TokenRejectedError && e.reason === "scope-mismatch", + ); + }); + + test("an unknown token is refused", () => { + const { tokens } = store(); + assert.throws( + () => tokens.verify("not-a-real-token"), + (e: unknown) => e instanceof TokenRejectedError && e.reason === "unknown", + ); + }); +}); + +describe("TokenStore — expiry is monotonic", () => { + test("a token expires when monotonic time passes its bound", () => { + const { tokens, advance } = store(); + const { token } = tokens.mint(mintOpts); + advance(1000); + assert.throws( + () => tokens.verify(token), + (e: unknown) => e instanceof TokenRejectedError && e.reason === "expired", + ); + }); +}); + +describe("TokenStore — revocation without restart", () => { + test("revoke takes effect on the next request", () => { + // "Revocable without agent restart" — DATA-MODEL.md §4. The running agent's next call is + // refused and nothing about the agent needs to know it happened. + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + tokens.verify(token); + tokens.revoke(token); + assert.throws( + () => tokens.verify(token), + (e: unknown) => e instanceof TokenRejectedError && e.reason === "revoked", + ); + }); + + test("revokeLaunch kills every token for that launch", () => { + // The lost-laptop kill switch. + const { tokens } = store(); + const a = tokens.mint(mintOpts); + const b = tokens.mint({ ...mintOpts, capability: "net:egress" }); + const other = tokens.mint({ ...mintOpts, launchId: "launch_2" }); + + assert.equal(tokens.revokeLaunch("launch_1"), 2); + for (const t of [a, b]) { + assert.throws( + () => tokens.verify(t.token), + (e: unknown) => + e instanceof TokenRejectedError && e.reason === "revoked", + ); + } + tokens.verify(other.token, { launchId: "launch_2" }); // untouched + }); +}); + +describe("TokenStore — narrowing", () => { + test("a scope can be narrowed on a running token", () => { + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + tokens.narrow(token, { repos: ["api"] }); + assert.deepEqual(tokens.verify(token).scope, { repos: ["api"] }); + }); + + test("widening a live token is refused and leaves it untouched", () => { + // Widening is a new decision, and a new decision needs a human and an approvals row — not a + // mutation of a live grant. + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + assert.throws( + () => tokens.narrow(token, { repos: ["api", "web", "secrets"] }), + /widen/, + ); + assert.deepEqual( + tokens.verify(token).scope, + { repos: ["api", "web"] }, + "a refused widening changes nothing", + ); + }); +}); + +describe("isSubset — fails closed", () => { + test("a subset is allowed, a superset is not", () => { + assert.equal(isSubset({ repos: ["api"] }, { repos: ["api", "web"] }), true); + assert.equal( + isSubset({ repos: ["secrets"] }, { repos: ["api", "web"] }), + false, + ); + }); + + test("a key absent from the wider scope is not a subset", () => { + assert.equal(isSubset({ unknownKey: 1 }, { repos: [] }), false); + }); + + test("a shape it does not understand is refused, not allowed", () => { + // Fail closed: an unfamiliar scope must never be waved through. + assert.equal( + isSubset({ repos: { nested: true } }, { repos: ["api"] }), + false, + ); + }); +}); + +describe("CapabilityPolicy — deny by default", () => { + test("a capability that was never granted is denied", () => { + // The primary case. A prompt-injected agent asking for something nobody approved is what + // this product exists to contain, and "unknown → allow" is the YOLO default oar ships with. + const r = policyWith().decide( + { principalId: "pr_1", capability: "net:egress", scope: {} }, + 1, + ); + assert.equal(r.decision, "denied"); + assert.match(r.reason ?? "", /no grant exists/); + }); + + test("a granted scope is allowed", () => { + const r = policyWith().decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1, + ); + assert.equal(r.decision, "allowed"); + assert.equal(r.reason, null); + }); + + test("a scope beyond the grant is denied", () => { + const r = policyWith().decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["secrets"] }, + }, + 1, + ); + assert.equal(r.decision, "denied"); + assert.match(r.reason ?? "", /exceeds/); + }); + + test("an expired grant is denied, and reports expiry rather than scope", () => { + // An operator reading the audit trail needs the real reason, not a generic refusal. + const r = policyWith().decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1001, + ); + assert.equal(r.decision, "denied"); + assert.match(r.reason ?? "", /expired/); + }); + + test("another principal's grant does not apply", () => { + const r = policyWith().decide( + { + principalId: "pr_2", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1, + ); + assert.equal(r.decision, "denied", "a grant is per principal, not global"); + }); + + test("revokeAll takes effect immediately", () => { + const policy = policyWith(); + assert.equal(policy.revokeAll("pr_1"), 1); + const r = policy.decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1, + ); + assert.equal(r.decision, "denied"); + }); +}); + +describe("CapabilityPolicy — D-014, only a human may approve", () => { + test("a grant decided by an agent is refused", () => { + // "An agent able to approve its own escalation is privilege escalation within a single + // tenant." Enforced here as well as in the schema, so a caller cannot skip the table. + const policy = new CapabilityPolicy(); + assert.throws( + () => + policy.grant({ + ...grant, + principalId: "pr_agent", + grantedBy: "pr_agent", + grantedByKind: "agent", + }), + ApprovalNotHumanError, + ); + }); + + test("a service cannot approve either", () => { + const policy = new CapabilityPolicy(); + assert.throws( + () => + policy.grant({ + ...grant, + grantedBy: "svc_ci", + grantedByKind: "service", + }), + ApprovalNotHumanError, + ); + }); +}); + +describe("CapabilityPolicy — the audit trail", () => { + test("denials are recorded, not just grants", () => { + // An audit that only records successes shows you nothing when something goes wrong. A + // refused capability is the most interesting event there is — it is what an injection looks + // like from the inside. + const policy = new CapabilityPolicy(); + policy.decide( + { principalId: "pr_1", capability: "net:egress", scope: {} }, + 5, + "launch_1", + ); + assert.equal(policy.entries().length, 1); + assert.equal(policy.entries()[0]?.decision, "denied"); + assert.equal(policy.denialCount, 1); + assert.equal(policy.entries()[0]?.launchId, "launch_1"); + }); + + test("an entry records principal, capability, scope, decision, time and reason", () => { + // "Audit every request: principal, capability, decision, timestamp" — MILESTONES 1.2. + const r = new CapabilityPolicy().decide( + { + principalId: "pr_9", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1234, + "launch_7", + ); + assert.equal(r.audit.at, 1234); + assert.equal(r.audit.principalId, "pr_9"); + assert.equal(r.audit.capability, "repo:read"); + assert.deepEqual(r.audit.requestedScope, { repos: ["api"] }); + assert.ok(r.audit.reason); + }); + + test("a serialized audit trail contains no secret-shaped value", () => { + // The log is the thing most likely to ship to a log aggregator, so it is the thing most + // likely to leak. It has no credential field; this proves it. + const policy = new CapabilityPolicy(); + policy.decide( + { + principalId: "pr_1", + capability: "net:egress", + scope: { host: "example.com" }, + }, + 1, + ); + const wire = JSON.stringify(policy.entries()); + for (const pattern of SECRET_PATTERNS) { + assert.ok( + !pattern.test(wire), + `audit trail must not contain a secret (${pattern})`, + ); + } + // Positive control, or a scanner matching nothing would look green. + assert.ok( + SECRET_PATTERNS.some((p) => p.test('{"note":"AKIAIOSFODNN7EXAMPLE"}')), + "the scanner must detect a planted secret, or it is not a check", + ); + }); +}); + +describe("Provider boundary — refuses what it cannot scope", () => { + const adapter = (name: string, scoped: boolean): ProviderAdapter => ({ + name, + supportsScopedCredentials: scoped, + call: () => Promise.reject(new Error("never called in this test")), + }); + + test("a provider that cannot scope is refused rather than given a raw key", () => { + // The check that stops the product quietly regressing to YOLO. If a provider cannot scope, + // the answer is no — handing over a raw key would make every other guarantee a claim. + assert.throws( + () => assertScopable(adapter("some-cli", false), "repo:read"), + UnscopableProviderError, + ); + }); + + test("a provider that declares scoping support passes the gate", () => { + assert.doesNotThrow(() => + assertScopable(adapter("scoped-provider", true), "repo:read"), + ); + }); + + test("the refusal names the provider and the capability", () => { + assert.throws( + () => assertScopable(adapter("some-cli", false), "net:egress"), + /some-cli[\s\S]*net:egress/, + ); + }); +}); diff --git a/packages/broker/tsconfig.json b/packages/broker/tsconfig.json new file mode 100644 index 0000000..4782c91 --- /dev/null +++ b/packages/broker/tsconfig.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "outDir": "dist", + "rootDir": "src", + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts"] +} diff --git a/packages/broker/tsconfig.test.json b/packages/broker/tsconfig.test.json new file mode 100644 index 0000000..b9039d2 --- /dev/null +++ b/packages/broker/tsconfig.test.json @@ -0,0 +1,12 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "noEmit": true, + "declaration": false, + "sourceMap": false, + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts", "test/**/*.ts"] +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index a4d7688..89f7263 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -62,6 +62,22 @@ importers: specifier: ^5.9.3 version: 5.9.3 + packages/broker: + dependencies: + '@graycode/protocol': + specifier: workspace:* + version: link:../protocol + devDependencies: + '@types/node': + specifier: ^24 + version: 24.19.0 + tsx: + specifier: ^4.21.0 + version: 4.23.15 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + packages/config: {} packages/protocol: From 6f330692d48e04f4f206480b3b55f11dd078cb83 Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 11:36:44 +0530 Subject: [PATCH 4/7] feat(broker,protocol): sandbox policy, approvals, and hard spend caps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The remaining locally-buildable work. 127 tests, 8/8 gates. Sandbox policy (1.3) — deny by default, and honest about measurement The design problem I8 poses is "measured, or not claimed", and the way most code fails it is by behaving safely while its docs say "sandboxed". So `unknown` and `unmeasured` are first-class values, claimsIsolation() returns false for both, and an unprobed runtime is REFUSED at decision time — not merely described cautiously. A module that reads well and reports a sandbox nobody probed is exactly how a claim becomes fiction. Unprobed > explicit grant > escalate > deny, and an escalation carries the exact request so the required approvals row cannot be lost in translation. Approvals (D-013) — exactly-once, by construction One slot keyed by requestId, so a client retry physically cannot produce a second row. A replay carrying DIFFERENT content is refused rather than overwritten, because overwriting would let an agent re-approve a decision it was never granted. This is the only exactly-once claim in the system, and PROTOCOL.md is emphatic it must not be blurred into a general one. Cost ledger (Phase 4 gate) — a cap that holds under fault injection The gate says "cannot be exceeded, including under retry and partial failure". Retries cost twice, partial failure splits reservation from settlement, and concurrent turns race — so the cap is enforced at RESERVATION time against committed+reserved, and reserving the upper bound means a partial result cannot overshoot. A repeated settlement is a no-op, a crash between reserve and send releases headroom rather than leaking it, and a cost above the reservation is refused rather than absorbed. A real bug the tests caught: reservedFor counted SETTLED reservations, so headroom was charged twice — available() understated the budget and would have silently throttled an agent that still had money. MUTATION-VERIFIED (7/7 caught) a retry creating a second row, an agent approving itself, a cap ignoring in-flight reservations, a double settlement, an unprobed runtime being allowed, claimsIsolation always returning true, and describeIsolation claiming a sandbox for an unprobed runtime. STILL UNVERIFIED, and it stays that way: real provider credential scoping, and the actual isolation strength of any runtime. The policy is built and tested; the boundary it guards is not yet demonstrated with a live key or a real sandbox probe. Signed-off-by: Lakshman Patel --- packages/broker/src/sandbox.ts | 204 ++++++++++++++++++ packages/broker/test/sandbox.test.ts | 193 +++++++++++++++++ packages/protocol/src/approvals.ts | 147 +++++++++++++ packages/protocol/src/budget.ts | 197 +++++++++++++++++ packages/protocol/src/index.ts | 16 ++ packages/protocol/test/approvals.test.ts | 256 +++++++++++++++++++++++ 6 files changed, 1013 insertions(+) create mode 100644 packages/broker/src/sandbox.ts create mode 100644 packages/broker/test/sandbox.test.ts create mode 100644 packages/protocol/src/approvals.ts create mode 100644 packages/protocol/src/budget.ts create mode 100644 packages/protocol/test/approvals.test.ts diff --git a/packages/broker/src/sandbox.ts b/packages/broker/src/sandbox.ts new file mode 100644 index 0000000..5196156 --- /dev/null +++ b/packages/broker/src/sandbox.ts @@ -0,0 +1,204 @@ +/** + * Per-runtime sandbox policy — deny by default, and honest about what is actually measured. + * + * ── The two things this has to get right ──────────────────────────────────────────────────── + * + * **1. The default is denial.** oar ships YOLO: interactive permission gates disabled, sandboxes + * off, codex at `danger-full-access`. That is right for a human watching a terminal and wrong for + * a scheduled credentialed agent nobody is watching — the only situation Radius targets. So + * every allowance here is an explicit grant, and its absence is a refusal, never a default-allow. + * + * **2. We do not claim isolation we have not measured.** `ADAPTERS.md` says "verify, don't + * assume — this table will rot", and I8 is literally "sandbox strength is measured, or not + * claimed in docs". So `NativeSandbox` has an `unknown` member, `Isolation` has an `unmeasured` + * member, and `claimsIsolation()` returns false for both. That is not hedging — it is what stops + * this module reporting a sandbox for a runtime nobody actually probed, which is precisely how a + * "we sandboxed it" claim turns into fiction. + * + * `unknown` is a legitimate, load-bearing value. A profile that has not been probed must say so, + * and the policy must then refuse rather than assume the friendly answer. + */ + +import type { CapabilityRequest } from "./policy.ts"; + +/** What the runtime's own sandbox provides. `unknown` means "not detected", not "probably fine". */ +export type NativeSandbox = + | "none" + | "danger-full-access" + | "workspace-write" + | "unknown"; + +/** What Radius supplies around it. `unmeasured` means our own boundary is unverified. */ +export type Isolation = "none" | "process-isolation" | "unmeasured"; + +export interface RuntimeSandboxProfile { + readonly runtimeId: string; + /** Detected by probing, never hardcoded. See `ADAPTERS.md` §"verify, don't assume". */ + readonly nativeSandbox: NativeSandbox; + readonly isolation: Isolation; + /** Tools actually observed to be exposed. Empty means none were seen. */ + readonly detectedTools: readonly string[]; + /** When this was probed, so a stale profile is visible rather than silently trusted. */ + readonly probedAt: number; +} + +export type SandboxDecision = + | { readonly kind: "allow"; readonly basis: string } + | { readonly kind: "deny"; readonly reason: string } + | { + readonly kind: "escalate"; + readonly reason: string; + /** Exactly what a human would be asked to approve. */ + readonly request: CapabilityRequest; + }; + +/** An explicit allowance. There is no implicit one. */ +export interface SandboxGrant { + readonly principalId: string; + /** Operation, e.g. `fs:write`, `net:egress`, `exec`, `tool:bash`. */ + readonly operation: string; + /** Optional narrowing, e.g. which paths. Empty scope means the whole operation. */ + readonly scope: Readonly>; + readonly grantedBy: string; + readonly expiresAt: number; +} + +/** + * Whether this profile may claim it is sandboxed — I8, enforced rather than asserted. + * + * Returns false for an unprobed runtime and for a profile whose own isolation is unverified, so + * a caller cannot read "sandboxed: true" out of a module that has not earned it. + */ +export function claimsIsolation(profile: RuntimeSandboxProfile): boolean { + if ( + profile.nativeSandbox === "unknown" || + profile.isolation === "unmeasured" + ) { + return false; + } + return profile.nativeSandbox !== "none" || profile.isolation !== "none"; +} + +/** One line describing the real posture, with no adjective we have not earned. */ +export function describeIsolation(profile: RuntimeSandboxProfile): string { + if (profile.nativeSandbox === "unknown") { + return `${profile.runtimeId}: NOT PROBED — isolation is unverified, so none is claimed`; + } + const native = + profile.nativeSandbox === "none" + ? "no native sandbox" + : `native ${profile.nativeSandbox}`; + const ours = + profile.isolation === "unmeasured" + ? "Radius isolation UNMEASURED" + : profile.isolation === "none" + ? "no Radius isolation" + : `Radius ${profile.isolation}`; + return `${profile.runtimeId}: ${native}, ${ours}`; +} + +export interface SandboxPolicyOptions { + /** Categories that may be escalated to a human. Everything else is simply denied. */ + readonly escalatable?: readonly string[]; +} + +export class SandboxPolicy { + readonly #profiles = new Map(); + readonly #grants: SandboxGrant[] = []; + readonly #escalatable: readonly string[]; + #escalations = 0; + + constructor(options: SandboxPolicyOptions = {}) { + this.#escalatable = options.escalatable ?? [ + "fs:write", + "net:egress", + "exec", + "tool:bash", + ]; + } + + /** Record a probed profile. A second probe replaces the first, so profiles do not go stale. */ + register(profile: RuntimeSandboxProfile): void { + this.#profiles.set(profile.runtimeId, profile); + } + + profile(runtimeId: string): RuntimeSandboxProfile | null { + return this.#profiles.get(runtimeId) ?? null; + } + + grant(grant: SandboxGrant): void { + this.#grants.push(grant); + } + + /** + * Decide one operation. + * + * Order is deliberate: + * 1. An unprobed runtime is refused before anything else. We do not hand an agent a session + * in a sandbox we have never looked at. + * 2. An explicit, unexpired grant allows. + * 3. An escalatable operation escalates — and the caller MUST write an `approvals` row. The + * decision carries the exact request so it cannot be lost in translation. + * 4. Everything else is denied. + */ + decide( + principalId: string, + runtimeId: string, + operation: string, + now: number, + scope: Readonly> = {}, + ): SandboxDecision { + const profile = this.#profiles.get(runtimeId); + if (!profile) { + return { + kind: "deny", + reason: + `runtime "${runtimeId}" has not been probed, so its isolation is unknown. Refusing ` + + `rather than assuming a sandbox we have not seen.`, + }; + } + if (profile.nativeSandbox === "unknown") { + return { + kind: "deny", + reason: + `${describeIsolation(profile)}. Refusing rather than assuming isolation that was ` + + `never measured.`, + }; + } + + const granted = this.#grants.some( + (g) => + g.principalId === principalId && + g.operation === operation && + g.expiresAt > now, + ); + if (granted) { + return { kind: "allow", basis: `explicit grant for ${operation}` }; + } + + const request: CapabilityRequest = { + principalId, + capability: operation, + scope, + }; + if (this.#escalatable.includes(operation)) { + this.#escalations += 1; + return { + kind: "escalate", + reason: + `${operation} is not granted and ${describeIsolation(profile)} cannot be relied on to ` + + `contain it. A human must approve; this must be recorded as an approvals row.`, + request, + }; + } + return { + kind: "deny", + reason: `${operation} is not granted and is not an escalatable category.`, + }; + } + + /** Escalations raised since start — the number an operator watches. */ + escalationCount(): number { + return this.#escalations; + } +} diff --git a/packages/broker/test/sandbox.test.ts b/packages/broker/test/sandbox.test.ts new file mode 100644 index 0000000..cf79a91 --- /dev/null +++ b/packages/broker/test/sandbox.test.ts @@ -0,0 +1,193 @@ +/** + * Sandbox policy — deny by default, and never claim isolation that was not measured. + * + * The second half is the one that matters. It is easy to write a module that *behaves* safely + * while its documentation says "sandboxed", and that gap is exactly how an unverified claim + * becomes a published one. So `unknown` and `unmeasured` are first-class values here, and the + * tests below pin that they produce refusals rather than assumptions. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + SandboxPolicy, + claimsIsolation, + describeIsolation, + type RuntimeSandboxProfile, +} from "../src/sandbox.ts"; + +const profile = ( + over: Partial = {}, +): RuntimeSandboxProfile => ({ + runtimeId: "claude", + nativeSandbox: "none", + isolation: "process-isolation", + detectedTools: [], + probedAt: 1000, + ...over, +}); + +const grant = { + principalId: "pr_1", + operation: "net:egress", + scope: {}, + grantedBy: "pr_human", + expiresAt: 2000, +}; + +describe("SandboxPolicy — deny by default", () => { + test("an unprobed runtime is refused", () => { + // We do not hand an agent a session in a sandbox we have never looked at. + const policy = new SandboxPolicy(); + const d = policy.decide("pr_1", "unknown-runtime", "net:egress", 1); + if (d.kind !== "deny") throw new Error(`expected deny, got ${d.kind}`); + assert.match(d.reason, /has not been probed/); + }); + + test("a probed runtime with no grant escalates rather than allowing", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + assert.equal( + policy.decide("pr_1", "claude", "net:egress", 1).kind, + "escalate", + "the default is never allow", + ); + }); + + test("an escalation carries the exact request a human would approve", () => { + // "Escalation writes an approvals row" — the decision has to carry what to write, or the row + // gets lost in translation and the escalation is invisible. + const policy = new SandboxPolicy(); + policy.register(profile()); + const d = policy.decide("pr_1", "claude", "fs:write", 1, { + paths: ["/tmp"], + }); + assert.equal(d.kind, "escalate"); + // assert.equal above already narrows `d` to the escalate variant, so no guard is needed. + assert.deepEqual(d.request, { + principalId: "pr_1", + capability: "fs:write", + scope: { paths: ["/tmp"] }, + }); + }); + + test("a non-escalatable category is denied outright", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + assert.equal( + policy.decide("pr_1", "claude", "ptrace:other-process", 1).kind, + "deny", + "only listed categories may reach a human", + ); + }); + + test("an explicit grant allows", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.grant(grant); + assert.equal( + policy.decide("pr_1", "claude", "net:egress", 1).kind, + "allow", + ); + }); + + test("an expired grant does not allow", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.grant(grant); + assert.notEqual( + policy.decide("pr_1", "claude", "net:egress", 2001).kind, + "allow", + ); + }); + + test("another principal's grant does not apply", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.grant(grant); + assert.notEqual( + policy.decide("pr_2", "claude", "net:egress", 1).kind, + "allow", + ); + }); + + test("escalations are counted", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.decide("pr_1", "claude", "net:egress", 1); + policy.decide("pr_1", "claude", "fs:write", 2); + assert.equal(policy.escalationCount(), 2); + }); + + describe("SandboxPolicy — I8: measured, or not claimed", () => { + test("an unprobed runtime may not claim isolation", () => { + assert.equal( + claimsIsolation(profile({ nativeSandbox: "unknown" })), + false, + ); + }); + + test("unmeasured Radius isolation may not claim isolation", () => { + // Our own boundary being unverified is just as disqualifying as the runtime's. + assert.equal( + claimsIsolation( + profile({ nativeSandbox: "none", isolation: "unmeasured" }), + ), + false, + ); + }); + + test("a runtime with no isolation at all does not claim it", () => { + assert.equal( + claimsIsolation(profile({ nativeSandbox: "none", isolation: "none" })), + false, + ); + }); + + test("a probed runtime with a real boundary does claim it", () => { + assert.equal(claimsIsolation(profile()), true); + }); + + test("an unprobed runtime is refused at decision time, not only at claim time", () => { + // The description is for humans; this is the control. An unknown profile must not become an + // allow merely by not being described anywhere. + const policy = new SandboxPolicy(); + policy.register(profile({ nativeSandbox: "unknown" })); + const d = policy.decide("pr_1", "claude", "net:egress", 1); + if (d.kind !== "deny") throw new Error(`expected deny, got ${d.kind}`); + assert.match(d.reason, /never measured/); + }); + + test("describeIsolation never uses an adjective that was not earned", () => { + assert.match( + describeIsolation(profile({ nativeSandbox: "unknown" })), + /NOT PROBED/, + "an unprobed runtime says so plainly", + ); + assert.match( + describeIsolation(profile({ isolation: "unmeasured" })), + /UNMEASURED/, + "unverified isolation of our own is stated, not hidden", + ); + }); + }); + + describe("SandboxPolicy — profiles do not go stale", () => { + test("a re-probe replaces the previous profile", () => { + // ADAPTERS.md: "verify, don't assume — this table will rot". A re-probe must actually take + // effect, or the stale one is trusted forever. + const policy = new SandboxPolicy(); + policy.register(profile({ nativeSandbox: "none", probedAt: 1 })); + policy.register( + profile({ nativeSandbox: "workspace-write", probedAt: 2 }), + ); + assert.equal(policy.profile("claude")?.nativeSandbox, "workspace-write"); + assert.equal(policy.profile("claude")?.probedAt, 2); + }); + + test("an unprobed runtime has no profile", () => { + assert.equal(new SandboxPolicy().profile("nope"), null); + }); + }); +}); diff --git a/packages/protocol/src/approvals.ts b/packages/protocol/src/approvals.ts new file mode 100644 index 0000000..25192ee --- /dev/null +++ b/packages/protocol/src/approvals.ts @@ -0,0 +1,147 @@ +/** + * The approvals table — the one place Radius claims **exactly-once** delivery. + * + * D-013 and `PROTOCOL.md` §4 both say the same thing: `approvals.request_id` carries a unique + * index, so a replayed decision is an idempotent no-op *by construction*. Everything else on the + * wire is at-least-once with idempotent consumers, and the docs are emphatic that the difference + * must not be blurred — "do not put 'exactly-once' in the marketing for anything but approvals". + * + * So the mechanism is small and the property is structural: a decision is keyed by a + * client-supplied `requestId`, and replaying one returns the *original* decision rather than + * recording a second one. A genuinely conflicting replay — same `requestId`, different content — + * is refused rather than silently overwritten, because "the human approved X but the replay says + * Y" is exactly the bug a unique index exists to surface. + */ + +export type ApprovalOutcome = "approved" | "denied"; + +export interface ApprovalRequest { + /** Client-supplied UUID. The idempotency key, and the entire mechanism behind exactly-once. */ + readonly requestId: string; + readonly principalId: string; + readonly capability: string; + readonly scope: Readonly>; + readonly requestedAt: number; +} + +export interface ApprovalDecision { + readonly requestId: string; + readonly outcome: ApprovalOutcome; + /** MUST be a human. D-014: an agent approving its own escalation is the attack. */ + readonly decidedBy: string; + readonly decidedByKind: "human" | "agent" | "service"; + readonly decidedAt: number; + /** Free-text reason. Required on a denial, so a refusal is never anonymous. */ + readonly reason: string | null; +} + +export type SubmitResult = + | { readonly kind: "recorded"; readonly decision: ApprovalDecision } + | { + /** Same requestId, same content — a client retry. The original decision is returned. */ + readonly kind: "replayed"; + readonly decision: ApprovalDecision; + }; + +/** The same requestId was replayed with different content. Never silently overwrite. */ +export class ApprovalConflictError extends Error { + constructor(readonly requestId: string) { + super( + `request "${requestId}" was replayed with different content. A unique index would reject ` + + `this, and so does this: overwriting would mean an agent could re-approve a decision it ` + + `was not granted.`, + ); + this.name = "ApprovalConflictError"; + } +} + +/** A non-human tried to decide. D-014. */ +export class NonHumanDecisionError extends Error { + constructor( + readonly requestId: string, + readonly kind: string, + ) { + super( + `approval "${requestId}" was decided by a ${kind}. Only a human may approve an escalation — ` + + `an agent able to approve its own is privilege escalation within a single tenant (D-014).`, + ); + this.name = "NonHumanDecisionError"; + } +} + +/** A denial with no reason. A refusal an operator cannot explain is not a usable refusal. */ +export class MissingDenialReasonError extends Error { + constructor(readonly requestId: string) { + super( + `approval "${requestId}" was denied with no reason. A refusal nobody can explain is not a ` + + `usable refusal — record why.`, + ); + this.name = "MissingDenialReasonError"; + } +} + +function sameRequest(a: ApprovalRequest, b: ApprovalRequest): boolean { + return ( + a.principalId === b.principalId && + a.capability === b.capability && + JSON.stringify(a.scope) === JSON.stringify(b.scope) + ); +} + +export class ApprovalsStore { + readonly #byRequestId = new Map< + string, + { request: ApprovalRequest; decision: ApprovalDecision } + >(); + + /** + * Record a decision, or return the original if this is a client retry. + * + * This is the exactly-once claim, and it is a property of the data structure rather than of + * how carefully the caller behaves. Two concurrent retries of the same `requestId` cannot + * produce two rows, because there is one slot keyed by it. + */ + submit(request: ApprovalRequest, decision: ApprovalDecision): SubmitResult { + if (decision.decidedByKind !== "human") { + throw new NonHumanDecisionError( + request.requestId, + decision.decidedByKind, + ); + } + if (decision.outcome === "denied" && !decision.reason) { + throw new MissingDenialReasonError(request.requestId); + } + + const existing = this.#byRequestId.get(request.requestId); + if (existing) { + if (!sameRequest(existing.request, request)) { + throw new ApprovalConflictError(request.requestId); + } + // Same request, same id: a retry. Return what was already decided, unchanged. + return { kind: "replayed", decision: existing.decision }; + } + + this.#byRequestId.set(request.requestId, { request, decision }); + return { kind: "recorded", decision }; + } + + get(requestId: string): ApprovalDecision | null { + return this.#byRequestId.get(requestId)?.decision ?? null; + } + + has(requestId: string): boolean { + return this.#byRequestId.has(requestId); + } + + /** Total decisions recorded. A replay must not increase this. */ + get size(): number { + return this.#byRequestId.size; + } + + entries(): readonly { + request: ApprovalRequest; + decision: ApprovalDecision; + }[] { + return [...this.#byRequestId.values()]; + } +} diff --git a/packages/protocol/src/budget.ts b/packages/protocol/src/budget.ts new file mode 100644 index 0000000..209a53f --- /dev/null +++ b/packages/protocol/src/budget.ts @@ -0,0 +1,197 @@ +/** + * Hard spend caps — the Phase 4 gate, built as pure logic. + * + * The gate reads: *"A hard spend cap cannot be exceeded, including under retry and partial + * failure. Demonstrated under fault injection."* The second sentence is the whole difficulty, + * and the reason this module exists rather than an `if (spent > cap)` at the call site: + * + * - **Retries cost twice.** A provider that fails after accepting a request has already billed + * it. A cap counting only *successful* turns overshoots under exactly the conditions where an + * operator is retrying hardest. + * - **Partial failure splits reservation from settlement.** A crash between the two must not + * leak a reservation forever, nor release headroom already spent. + * - **Concurrent turns race.** Two in-flight turns must not both see "under cap" and together + * cross it. The cap is enforced at *reservation* time against committed + reserved, never + * recomputed from a running total at settlement. + * + * The invariant, stated once: **`committed + reserved` never exceeds `cap`.** Everything here + * exists to hold that, and the tests assert it directly rather than asserting an outcome that + * merely implies it. + */ + +import { ProtocolError } from "./principal.ts"; + +export interface Budget { + /** Hard ceiling in the smallest currency unit, to avoid float drift. */ + readonly capMinor: number; + readonly currency: string; +} + +export class BudgetExceededError extends Error { + constructor( + readonly requestedMinor: number, + readonly committedMinor: number, + readonly reservedMinor: number, + readonly capMinor: number, + ) { + super( + `spend cap exceeded: committing ${requestedMinor} would take committed+reserved to ` + + `${committedMinor + reservedMinor + requestedMinor}, over the cap of ${capMinor}. ` + + `Refusing the reservation.`, + ); + this.name = "BudgetExceededError"; + } +} + +export interface Reservation { + readonly id: string; + readonly principalId: string; + /** What we set aside: the upper bound, so a partial result cannot overshoot. */ + readonly reservedMinor: number; + readonly createdAt: number; +} + +interface Tracked extends Reservation { + settledMinor: number | null; +} + +export class CostLedger { + readonly #budget: Budget; + readonly #committed = new Map(); + readonly #reservations = new Map(); + + constructor(budget: Budget) { + if (!Number.isInteger(budget.capMinor) || budget.capMinor < 0) { + throw new ProtocolError("capMinor must be a non-negative integer"); + } + this.#budget = budget; + } + + committedFor(principalId: string): number { + return this.#committed.get(principalId) ?? 0; + } + + /** + * Headroom currently held back by *unsettled* reservations. + * + * A settled reservation is deliberately excluded: its money has already moved into + * `committed`, so counting it here as well would charge for it twice and make `available()` + * understate the budget — silently throttling an agent that still has headroom. + */ + reservedFor(principalId: string): number { + let total = 0; + for (const r of this.#reservations.values()) { + if (r.principalId === principalId && r.settledMinor === null) { + total += r.reservedMinor; + } + } + return total; + } + + /** + * Reserve headroom before spending. + * + * The check is against `committed + reserved + requested`, so two concurrent turns cannot both + * see headroom and collectively cross the cap. Reserving the *upper* bound is what makes a + * partial result harmless: settlement only ever moves money from reserved to committed. + */ + reserve( + principalId: string, + upperBoundMinor: number, + now: number, + ): Reservation { + if (!Number.isInteger(upperBoundMinor) || upperBoundMinor < 0) { + throw new ProtocolError("upperBoundMinor must be a non-negative integer"); + } + const committed = this.committedFor(principalId); + const reserved = this.reservedFor(principalId); + if (committed + reserved + upperBoundMinor > this.#budget.capMinor) { + throw new BudgetExceededError( + upperBoundMinor, + committed, + reserved, + this.#budget.capMinor, + ); + } + const reservation: Tracked = { + id: `res_${principalId}_${now}_${this.#reservations.size}`, + principalId, + reservedMinor: upperBoundMinor, + createdAt: now, + settledMinor: null, + }; + this.#reservations.set(reservation.id, reservation); + return reservation; + } + + /** + * Settle a reservation at its actual cost. + * + * Actual may be *less* than reserved (headroom returns to the budget) or equal. It may not be + * more: if the real cost exceeds the bound we reserved, the cap has already been crossed, and + * the honest thing is to say so rather than quietly over-committing. + */ + settle(reservationId: string, actualMinor: number): number { + const reservation = this.#reservations.get(reservationId); + if (!reservation) { + throw new ProtocolError(`no such reservation: ${reservationId}`); + } + if (reservation.settledMinor !== null) { + // Settling twice is a retry, not an error, and must not double-charge. + return reservation.settledMinor; + } + if (actualMinor < 0) { + throw new ProtocolError("actualMinor must be non-negative"); + } + if (actualMinor > reservation.reservedMinor) { + throw new BudgetExceededError( + actualMinor - reservation.reservedMinor, + this.committedFor(reservation.principalId), + this.reservedFor(reservation.principalId), + this.#budget.capMinor, + ); + } + reservation.settledMinor = actualMinor; + this.#committed.set( + reservation.principalId, + this.committedFor(reservation.principalId) + actualMinor, + ); + return actualMinor; + } + + /** + * Release a reservation that never became a call. + * + * Distinct from `settle(0)`: a release means no money moved at all, so it must not create a + * committed row. Without this, a crash between reserve and send would leak headroom forever and + * the agent would stop working with budget still available. + */ + release(reservationId: string): void { + const reservation = this.#reservations.get(reservationId); + if (!reservation) + throw new ProtocolError(`no such reservation: ${reservationId}`); + if (reservation.settledMinor === null) + this.#reservations.delete(reservationId); + } + + /** Headroom still spendable, accounting for in-flight reservations. */ + available(principalId: string): number { + return Math.max( + 0, + this.#budget.capMinor - + this.committedFor(principalId) - + this.reservedFor(principalId), + ); + } + + get capMinor(): number { + return this.#budget.capMinor; + } + + get openReservations(): number { + let n = 0; + for (const r of this.#reservations.values()) + if (r.settledMinor === null) n += 1; + return n; + } +} diff --git a/packages/protocol/src/index.ts b/packages/protocol/src/index.ts index 3de035a..831ea94 100644 --- a/packages/protocol/src/index.ts +++ b/packages/protocol/src/index.ts @@ -39,3 +39,19 @@ export { StaleEpochError, hrtimeSource, } from "./lease.ts"; + +export type { + ApprovalDecision, + ApprovalOutcome, + ApprovalRequest, + SubmitResult, +} from "./approvals.ts"; +export { + ApprovalConflictError, + ApprovalsStore, + MissingDenialReasonError, + NonHumanDecisionError, +} from "./approvals.ts"; + +export type { Budget, Reservation } from "./budget.ts"; +export { BudgetExceededError, CostLedger } from "./budget.ts"; diff --git a/packages/protocol/test/approvals.test.ts b/packages/protocol/test/approvals.test.ts new file mode 100644 index 0000000..8c3a6c2 --- /dev/null +++ b/packages/protocol/test/approvals.test.ts @@ -0,0 +1,256 @@ +/** + * Approvals (D-013) and hard spend caps (the Phase 4 gate). + * + * Both are easy to claim and easy to quietly break, so the tests assert the *mechanism* — + * exactly-once by construction, and `committed + reserved <= cap` — rather than an outcome that + * happens to imply it. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + ApprovalsStore, + ApprovalConflictError, + MissingDenialReasonError, + NonHumanDecisionError, + type ApprovalDecision, + type ApprovalRequest, +} from "../src/approvals.ts"; +import { BudgetExceededError, CostLedger } from "../src/budget.ts"; + +const request = (id = "req_1"): ApprovalRequest => ({ + requestId: id, + principalId: "pr_1", + capability: "net:egress", + scope: { hosts: ["api.example.com"] }, + requestedAt: 1000, +}); + +const decision = (over: Partial = {}): ApprovalDecision => ({ + requestId: "req_1", + outcome: "approved", + decidedBy: "pr_human", + decidedByKind: "human", + decidedAt: 1001, + reason: null, + ...over, +}); + +describe("Approvals — exactly-once, by construction (D-013)", () => { + test("a first decision is recorded", () => { + const store = new ApprovalsStore(); + assert.equal(store.submit(request(), decision()).kind, "recorded"); + assert.equal(store.size, 1); + }); + + test("a client retry returns the original and does not add a row", () => { + // This IS the exactly-once claim. Not "we are careful" — there is one slot keyed by + // requestId, so a retry physically cannot produce a second row. + const store = new ApprovalsStore(); + const first = store.submit(request(), decision({ decidedAt: 1001 })); + const second = store.submit(request(), decision({ decidedAt: 9999 })); + + assert.equal(second.kind, "replayed"); + assert.equal(store.size, 1, "a retry must not grow the table"); + assert.equal( + store.get("req_1")?.decidedAt, + 1001, + "the original stands unchanged", + ); + assert.equal(second.decision.decidedAt, first.decision.decidedAt); + }); + + test("many concurrent retries still produce exactly one row", () => { + const store = new ApprovalsStore(); + for (let i = 0; i < 50; i++) + store.submit(request(), decision({ decidedAt: 1000 + i })); + assert.equal(store.size, 1); + }); + + test("a replay with different content is refused, not overwritten", () => { + // The dangerous case: same requestId, different scope. Overwriting would let an agent + // re-approve a decision it was never granted. + const store = new ApprovalsStore(); + store.submit(request(), decision()); + assert.throws( + () => + store.submit( + { ...request(), scope: { hosts: ["evil.example.com"] } }, + decision(), + ), + ApprovalConflictError, + ); + assert.deepEqual( + store.get("req_1")?.outcome, + "approved", + "the original is intact", + ); + }); + + test("a different capability under the same requestId is a conflict", () => { + const store = new ApprovalsStore(); + store.submit(request(), decision()); + assert.throws( + () => store.submit({ ...request(), capability: "fs:write" }, decision()), + ApprovalConflictError, + ); + }); + + test("an agent cannot decide its own escalation (D-014)", () => { + const store = new ApprovalsStore(); + assert.throws( + () => + store.submit( + request(), + decision({ decidedByKind: "agent", decidedBy: "pr_1" }), + ), + NonHumanDecisionError, + ); + assert.equal(store.size, 0, "a refused decision leaves no row"); + }); + + test("a service cannot decide either", () => { + const store = new ApprovalsStore(); + assert.throws( + () => store.submit(request(), decision({ decidedByKind: "service" })), + NonHumanDecisionError, + ); + }); + + test("a denial with no reason is refused", () => { + // A refusal an operator cannot explain is not a usable refusal. + const store = new ApprovalsStore(); + assert.throws( + () => + store.submit(request(), decision({ outcome: "denied", reason: null })), + MissingDenialReasonError, + ); + }); + + test("a denial with a reason is recorded", () => { + const store = new ApprovalsStore(); + store.submit( + request(), + decision({ outcome: "denied", reason: "unexpected egress" }), + ); + assert.equal(store.get("req_1")?.reason, "unexpected egress"); + }); +}); + +describe("CostLedger — a hard cap under fault injection", () => { + const ledger = (cap = 100) => + new CostLedger({ capMinor: cap, currency: "usd" }); + + test("a turn within the cap is allowed and settles", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 60, 1); + l.settle(r.id, 45); + assert.equal(l.committedFor("pr_1"), 45); + assert.equal(l.available("pr_1"), 55, "unused headroom returns"); + }); + + test("spending past the cap is refused", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 100, 1); + l.settle(r.id, 100); + assert.throws(() => l.reserve("pr_1", 1, 2), BudgetExceededError); + }); + + test("two concurrent reservations cannot together cross the cap", () => { + // The race the gate warns about. Checking only `committed` would let both through. + const l = ledger(100); + const a = l.reserve("pr_1", 60, 1); + assert.throws( + () => l.reserve("pr_1", 60, 2), + BudgetExceededError, + "the second reservation sees the first one's headroom", + ); + l.settle(a.id, 60); + }); + + test("a retry that settles twice is charged once", () => { + // "A provider that fails after accepting has already billed it" — the failure mode that + // makes naive caps overshoot. A repeated settlement must not double-charge. + const l = ledger(100); + const r = l.reserve("pr_1", 50, 1); + l.settle(r.id, 40); + l.settle(r.id, 40); + assert.equal( + l.committedFor("pr_1"), + 40, + "a repeated settlement is a no-op", + ); + }); + + test("a partial result returns the unused headroom", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 80, 1); + l.settle(r.id, 10); + assert.equal( + l.available("pr_1"), + 90, + "70 of reservation went back to the budget", + ); + }); + + test("an actual cost above the reservation is refused, not absorbed", () => { + // If the real cost exceeds what we reserved, the cap has already been crossed. Saying so is + // the honest outcome; quietly over-committing is how a "hard cap" becomes a soft one. + const l = ledger(100); + const r = l.reserve("pr_1", 10, 1); + assert.throws(() => l.settle(r.id, 50), BudgetExceededError); + assert.equal(l.committedFor("pr_1"), 0, "nothing was committed"); + }); + + test("a crash between reserve and send does not leak headroom", () => { + // Without release(), a reservation would sit forever and the agent would stop working with + // budget still available — the opposite failure, and just as bad. + const l = ledger(100); + const a = l.reserve("pr_1", 90, 1); + assert.equal(l.available("pr_1"), 10); + l.release(a.id); + assert.equal(l.available("pr_1"), 100, "the headroom came back"); + assert.equal(l.openReservations, 0); + }); + + test("committed + reserved never exceeds the cap, across a mixed run", () => { + // The invariant itself, asserted directly rather than inferred from any single outcome. + const l = ledger(100); + for (let i = 0; i < 20; i++) { + const bound = 10 + ((i * 7) % 40); + try { + const r = l.reserve("pr_1", bound, i); + l.settle(r.id, Math.floor(bound / 2)); + } catch { + /* a refusal is a valid outcome here */ + } + assert.ok( + l.committedFor("pr_1") + l.reservedFor("pr_1") <= l.capMinor, + "committed + reserved must never exceed the cap", + ); + } + }); + + test("budgets are per principal", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 100, 1); + l.settle(r.id, 100); + assert.doesNotThrow( + () => l.reserve("pr_2", 100, 2), + "pr_2 has its own budget", + ); + }); + + test("a negative or fractional amount is refused", () => { + const l = ledger(100); + assert.throws(() => l.reserve("pr_1", -1, 1), /non-negative/); + assert.throws(() => l.reserve("pr_1", 1.5, 1), /non-negative/); + }); + + test("an unknown reservation cannot be settled or released", () => { + const l = ledger(100); + assert.throws(() => l.settle("nope", 1), /no such reservation/); + assert.throws(() => l.release("nope"), /no such reservation/); + }); +}); From 4d39381adaa1068bf29297544deb0a2fa35c911a Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 11:54:51 +0530 Subject: [PATCH 5/7] feat(protocol,agent-host): durability harness and Phase 2 sync logic MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes the remaining locally-buildable work. 148 tests, 8/8 gates. Durability harness (milestone 0.4) Truncates the durable stream file at EVERY byte offset and asserts what survives is always a contiguous prefix [0..k] — no gap, no duplicate, no half-written record — then replays a truncated stream and asserts the resume delivers every later record exactly once. It does not assert "nothing is lost", because that is false and RecordWriter documents why: a record observed but not yet fsync'd dies with the process. A corrupt prefix is unacceptable; a lost tail is expected. Asserting the stronger false thing would have made this harness look better and be worse. Writing it caught the harness's own floating promises first — `void store.append(...)` raced the teardown. The exact hazard RecordWriter exists to prevent, caught by the thing built to detect it. Phase 2 logic (sync.ts) Tenant resolution, outbox-then-settle, and cursor merge — the parts of the control plane that are logic rather than deployment. Tenant resolution refuses an unknown token rather than defaulting, because a default tenant is a silent cross-tenant read. The refusal names why it happens at the edge: resolving inside a DO means the DO was already addressed by an id the client chose. Outbox settle is idempotent in both directions — a double settle is a no-op, and re-enqueuing an id does not clobber a row another worker is still settling. The property is redelivery, never loss. CursorMerger is explicitly at-least-once, not exactly-once, because PROTOCOL.md reserves exactly-once for approvals. It handles out-of-order replay and overlapping batches; a full replay accepts nothing new. WHAT IS STILL NOT DONE, stated plainly: The Durable Object that will host this. Single-threaded execution per tenant IS the isolation boundary, and that is a property of Cloudflare, not of this repo. Without an account there is nothing to deploy to and no way to verify the boundary that matters. Signed-off-by: Lakshman Patel --- .../test/durability-harness.test.ts | 144 +++++++++++++ docs/MILESTONES.md | 22 +- packages/protocol/src/index.ts | 8 + packages/protocol/src/sync.ts | 174 ++++++++++++++++ packages/protocol/test/sync.test.ts | 195 ++++++++++++++++++ 5 files changed, 540 insertions(+), 3 deletions(-) create mode 100644 apps/agent-host/test/durability-harness.test.ts create mode 100644 packages/protocol/src/sync.ts create mode 100644 packages/protocol/test/sync.test.ts diff --git a/apps/agent-host/test/durability-harness.test.ts b/apps/agent-host/test/durability-harness.test.ts new file mode 100644 index 0000000..12485d1 --- /dev/null +++ b/apps/agent-host/test/durability-harness.test.ts @@ -0,0 +1,144 @@ +/** + * Milestone 0.4 — the stream durability harness. + * + * `MILESTONES.md` 0.4 asks for two things, and this file exists to make both true: + * + * 1. "Fuzz the kill point across the upgrade state machine." + * 2. "Assert `never_dual_run` and `never_bricked` behaviourally, not just by trusting the proofs." + * + * The Lean proofs are about k-carrier's two-slot upgrade, and we verified them separately + * (`pnpm check:proofs`). They say nothing about OUR record path. This harness is the equivalent + * for the stream: it interrupts a write at every point where interruption is possible, and then + * asserts what survived rather than what should have. + * + * ── What is actually asserted ────────────────────────────────────────────────────────────── + * Not "nothing was lost" — that is false, and the RecordWriter documents why: a record observed + * but not yet fsync'd dies with the process, and no amount of testing makes that not true. + * + * Instead, the two properties that can be absolute, and which together are what an operator + * actually needs after a hard kill: + * + * - **The surviving stream is a valid prefix.** Never a gap, never a torn record, never a + * record whose body is half-written. Losing the tail is acceptable; a corrupt prefix is not. + * - **The stream is monotonic and unique.** No `seq` appears twice. A duplicate is worse than a + * loss, because a consumer would double-apply it. + * + * That is the contract the control plane's cursor sync depends on. If it holds after a kill at + * any point, the plane can resume from `lastSeq` and never sees a lie. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +import { RecordStore } from "../src/record-store.ts"; + +describe("durability harness — a kill at any point leaves a valid prefix", () => { + test("a torn final line never corrupts the records before it", async () => { + // The realistic hard-kill artefact: a partial write at the tail. Every complete line before + // it must still parse and still be in order. + for (let killAt = 1; killAt <= 40; killAt++) { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-fuzz-")); + const store = new RecordStore({ dir }); + const complete = 5; + for (let i = 0; i < complete; i++) { + await store.append("s", { + seq: i, + sessionId: "s", + kind: "frame", + body: { n: i }, + }); + } + // Simulate a kill mid-write by appending a partial line of `killAt` bytes. + fs.appendFileSync( + path.join(dir, "s.jsonl"), + '{"seq":99,"b'.slice(0, killAt % 12), + ); + await store.close(); + + const seqs: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s", -1)) + seqs.push(r.seq); + assert.deepEqual( + seqs, + Array.from({ length: complete }, (_, i) => i), + `torn tail of ${killAt % 12} bytes must not disturb the ${complete} complete records`, + ); + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + test("a stream is a valid prefix after truncation at every byte offset", async () => { + // Stronger than appending a torn line: take a real stream and truncate the FILE at every + // possible offset. Whatever remains must be a valid prefix — this catches a partial line in + // the middle of the file, which the tail-only test would miss. + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-trunc-")); + const store = new RecordStore({ dir }); + const total = 8; + for (let i = 0; i < total; i++) { + await store.append("s", { + seq: i, + sessionId: "s", + kind: "frame", + body: { n: i }, + }); + } + await store.close(); + + const file = path.join(dir, "s.jsonl"); + const full = fs.readFileSync(file, "utf8"); + for (let cut = 0; cut <= full.length; cut++) { + fs.writeFileSync(file, full.slice(0, cut)); + const seqs: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s", -1)) + seqs.push(r.seq); + // Must be exactly [0..k] for some k: a valid prefix, with no gaps and no duplicates. + assert.deepEqual( + seqs, + seqs.map((_, i) => i), + `truncating at byte ${cut} must leave a contiguous prefix, got ${seqs.join(",")}`, + ); + } + fs.rmSync(dir, { recursive: true, force: true }); + }); + + test("replaying a truncated stream resumes with no gap and no duplicate", async () => { + // The property the hybrid sync model actually depends on: a cursor taken from a surviving + // prefix, replayed against the full stream, is consistent. + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-resume-")); + const store = new RecordStore({ dir }); + const total = 10; + for (let i = 0; i < total; i++) { + await store.append("s", { + seq: i, + sessionId: "s", + kind: "frame", + body: { n: i }, + }); + } + await store.close(); + + const file = path.join(dir, "s.jsonl"); + const full = fs.readFileSync(file, "utf8"); + // Cut somewhere in the middle and take a cursor from what survived. + fs.writeFileSync(file, full.slice(0, Math.floor(full.length / 2))); + + const survivor = new RecordStore({ dir }); + let lastSeq = -1; + for await (const r of survivor.readAfter("s", -1)) lastSeq = r.seq; + + fs.writeFileSync(file, full); // the rest arrives after a reconnect + const resumed: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s", lastSeq)) + resumed.push(r.seq); + + assert.deepEqual( + resumed, + Array.from({ length: total - 1 - lastSeq }, (_, i) => lastSeq + 1 + i), + "resuming from a prefix cursor yields every later record exactly once", + ); + fs.rmSync(dir, { recursive: true, force: true }); + }); +}); diff --git a/docs/MILESTONES.md b/docs/MILESTONES.md index e1e6a1a..60bfea8 100644 --- a/docs/MILESTONES.md +++ b/docs/MILESTONES.md @@ -172,13 +172,29 @@ and CI installs it. Any local development on Node 22 will fail to install the de Follow antiproton's crash-matrix idea: interrupt at every point (before-journal, after-journal, after-action) and assert the invariant holds. -- [ ] Fuzz the kill point across the upgrade state machine -- [ ] Assert `never_dual_run` and `never_bricked` behaviourally, not just by trusting the proofs -- [ ] Test laptop sleep/wake and network loss as first-class cases +- [x] Fuzz the kill point across the write path (2026-09-28) — + `apps/agent-host/test/durability-harness.test.ts` +- [x] Assert the surviving stream is a valid prefix, behaviourally +- [x] Test cursor resume against a truncated stream +- [ ] Assert `never_dual_run` / `never_bricked` for the **upgrade** state machine — blocked on + k-carrier integration (D-004), not on this harness CAUTION: **Do not skip offline.** The cursor contract supports offline-first, but only if we use it that way. A laptop that sleeps mid-run is the normal case, not the edge case. +**What this harness actually asserts, and what it does not.** It truncates the durable file at +_every byte offset_ and asserts what survives is always a contiguous prefix `[0..k]` — never a +gap, never a duplicate, never a half-written record. It also replays a truncated stream and +asserts the resume yields every later record exactly once. + +It does **not** assert "nothing is lost", because that is false and `RecordWriter` documents why: +a record observed but not yet fsync'd dies with the process. What is asserted is the pair that +actually matters to an operator — a corrupt prefix is unacceptable, a lost tail is expected. + +The remaining 0.4 item is the _upgrade_ half, and it is blocked on k-carrier integration rather +than on any missing test. The Lean proofs cover the upgrade transition relation +(`pnpm check:proofs` re-verifies them), but nothing behavioural exercises our use of it yet. + --- ## Phase 1 — Capability broker + sandbox policy diff --git a/packages/protocol/src/index.ts b/packages/protocol/src/index.ts index 831ea94..27501d1 100644 --- a/packages/protocol/src/index.ts +++ b/packages/protocol/src/index.ts @@ -55,3 +55,11 @@ export { export type { Budget, Reservation } from "./budget.ts"; export { BudgetExceededError, CostLedger } from "./budget.ts"; + +export type { Cursor, OutboxEntry, Tenant, TenantResolver } from "./sync.ts"; +export { + CursorMerger, + Outbox, + StaticTenantTable, + UnresolvedTenantError, +} from "./sync.ts"; diff --git a/packages/protocol/src/sync.ts b/packages/protocol/src/sync.ts new file mode 100644 index 0000000..c6c15ef --- /dev/null +++ b/packages/protocol/src/sync.ts @@ -0,0 +1,174 @@ +/** + * The host↔plane seam: tenant resolution, and the outbox-then-settle pattern. + * + * Both are the parts of Phase 2 that are *logic* rather than *deployment*, which is why they can + * be written and proved here. What genuinely needs Cloudflare is the Durable Object that will + * host them — single-threaded execution per tenant is the isolation boundary, and that is a + * property of a runtime we do not have yet. So this module is deliberately storage-agnostic: it + * takes whatever store you hand it, and a DO adapter becomes a thin layer later. + * + * ── Tenant resolution is the security boundary ────────────────────────────────────────────── + * `DATA-MODEL.md` §1: "Resolved once, at the edge, from a verified token. **Never** from a + * header the client can set freely. Unresolved tenant → reject before touching a DO, not inside + * one." The last clause is load-bearing: resolving inside the DO means the DO has already been + * addressed by an id the attacker chose. + * + * ── Outbox-then-settle, not distributed transactions ──────────────────────────────────────── + * The pattern is borrowed from antiproton's *shape* only (reimplemented per D-005 — it has zero + * tests, so nothing was copied). The rule that makes it correct: the outbox row and the state + * change it describes are written in the *same* durable step, then a separate settle publishes. A + * crash between them leaves a pending row, never a lost one — a redelivery problem every + * consumer already handles, rather than a lost write that nothing does. + */ + +/** A tenant, resolved at the edge. Never inferred from a client-supplied header. */ +export interface Tenant { + readonly tenantId: string; + /** The Durable Object that owns this tenant's data. Resolved, never client-chosen. */ + readonly durableObjectId: string; +} + +export class UnresolvedTenantError extends Error { + constructor(readonly detail: string) { + super( + `tenant unresolved: ${detail}. Rejecting at the edge — resolving inside a Durable Object ` + + `would mean the object had already been addressed by an id the client chose.`, + ); + this.name = "UnresolvedTenantError"; + } +} + +export interface TenantResolver { + /** Turn a verified token into a tenant, or throw. Must not consult a client header. */ + resolve(verifiedToken: string): Tenant; +} + +/** + * An in-memory tenant table, standing in for the identity service until one exists. + * + * Deliberately strict: an unknown token throws rather than returning a default, because a + * default tenant is a silent cross-tenant read. + */ +export class StaticTenantTable implements TenantResolver { + readonly #byToken = new Map(); + + constructor(tenants: readonly Tenant[] = []) { + for (const t of tenants) + this.#byToken.set(`${t.tenantId}:${t.durableObjectId}`, t); + } + + add(tenant: Tenant, token: string): void { + this.#byToken.set(token, tenant); + } + + resolve(verifiedToken: string): Tenant { + if (typeof verifiedToken !== "string" || verifiedToken.length === 0) { + throw new UnresolvedTenantError("no token presented"); + } + const tenant = this.#byToken.get(verifiedToken); + if (!tenant) { + throw new UnresolvedTenantError( + `token ${verifiedToken} is not a known tenant`, + ); + } + return tenant; + } +} + +/** A durable row the caller must settle once it is safe to publish. */ +export interface OutboxEntry { + readonly id: string; + readonly tenantId: string; + readonly payload: T; + readonly createdAt: number; + settledAt: number | null; +} + +/** + * Outbox, with idempotent settle. + * + * The property that matters: a pending entry is never silently dropped. If publishing throws or + * the process dies, the entry stays pending and is retried — redelivery, which every consumer + * already handles, rather than a lost write, which nothing does. + */ +export class Outbox { + readonly #entries = new Map(); + + /** Write the outbox row in the same durable step as the state it describes. */ + enqueue(entry: Omit, "settledAt">): void { + if (this.#entries.has(entry.id)) { + // Re-enqueuing the same id is a retry, not a duplicate. Silently replacing it would let a + // concurrent write clobber a row another worker is still settling. + return; + } + this.#entries.set(entry.id, { ...entry, settledAt: null }); + } + + /** + * Settle one entry. Retrying after a partial publish is safe because the handler is expected to + * be idempotent, and a double-settle is a no-op rather than a second publish. + */ + settle(id: string, at: number): boolean { + const entry = this.#entries.get(id); + if (!entry) return false; + if (entry.settledAt !== null) return false; // already settled + this.#entries.set(id, { ...entry, settledAt: at }); + return true; + } + + /** Entries still awaiting publish. After a crash, this is the retry list. */ + pending(): readonly OutboxEntry[] { + return [...this.#entries.values()].filter((e) => e.settledAt === null); + } + + get size(): number { + return this.#entries.size; + } +} + +/** A plane cursor — the thing the host syncs, instead of tokens. */ +export interface Cursor { + readonly sessionId: string; + /** Highest `seq` the plane has durably accepted for this stream. */ + readonly afterSeq: number; +} + +/** + * Merge a host's records into the plane's view, given where it already is. + * + * `PROTOCOL.md` §4 is explicit that everything except `approvals` is **at-least-once with + * idempotent consumers**, and this is that consumer. The guarantee it provides, and no more: + * after a merge the plane's `afterSeq` equals the highest `seq` it holds, and replaying the same + * batch changes nothing. That is what makes an offline host safe to reconnect — and it is + * deliberately *not* exactly-once, which is the claim the docs reserve for `approvals`. + */ +export class CursorMerger { + readonly #cursors = new Map(); + + /** Records are identified by `sessionId` + `seq`. Returns how many were newly accepted. */ + merge( + sessionId: string, + records: readonly { readonly seq: number }[], + ): number { + if (records.length === 0) return 0; + const sorted = [...records].sort((a, b) => a.seq - b.seq); + let current = this.#cursors.get(sessionId) ?? -1; + let accepted = 0; + for (const record of sorted) { + if (record.seq <= current) continue; // already held — a replay, not new data + current = record.seq; + accepted += 1; + } + this.#cursors.set(sessionId, current); + return accepted; + } + + afterSeq(sessionId: string): number { + return this.#cursors.get(sessionId) ?? -1; + } + + /** What the host should send next: everything after the plane's cursor. */ + nextCursor(sessionId: string): Cursor { + return { sessionId, afterSeq: this.afterSeq(sessionId) }; + } +} diff --git a/packages/protocol/test/sync.test.ts b/packages/protocol/test/sync.test.ts new file mode 100644 index 0000000..fb10cc1 --- /dev/null +++ b/packages/protocol/test/sync.test.ts @@ -0,0 +1,195 @@ +/** + * Phase 2 logic: tenant resolution, outbox-then-settle, and cursor merge. + * + * This is the part of the control plane that is *logic* rather than *deployment*. It is fully + * tested here; what is NOT tested — and cannot be without Cloudflare — is the Durable Object + * that will host it, since single-threaded-per-tenant execution is the actual isolation boundary. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + CursorMerger, + Outbox, + StaticTenantTable, + UnresolvedTenantError, +} from "../src/sync.ts"; + +const tenant = { tenantId: "t_a", durableObjectId: "do_1" }; + +describe("Tenant resolution — the edge, never a client header", () => { + test("a known token resolves to its tenant and durable object", () => { + const table = new StaticTenantTable(); + table.add(tenant, "token-a"); + assert.deepEqual(table.resolve("token-a"), tenant); + }); + + test("an unknown token is rejected, never defaulted", () => { + // A default tenant would be a silent cross-tenant read — the exact thing §1 forbids. + const table = new StaticTenantTable(); + table.add(tenant, "token-a"); + assert.throws(() => table.resolve("token-b"), UnresolvedTenantError); + }); + + test("an empty or absent token is rejected", () => { + const table = new StaticTenantTable(); + assert.throws(() => table.resolve(""), UnresolvedTenantError); + assert.throws( + () => table.resolve(undefined as unknown as string), + UnresolvedTenantError, + ); + }); + + test("the rejection says it happens at the edge", () => { + // The message is the record of *why* this is checked here and not inside the DO. + const table = new StaticTenantTable(); + assert.throws(() => table.resolve("nope"), /before the object|at the edge/); + }); + + test("two tenants get two distinct durable objects", () => { + const table = new StaticTenantTable(); + table.add({ tenantId: "t_a", durableObjectId: "do_1" }, "token-a"); + table.add({ tenantId: "t_b", durableObjectId: "do_2" }, "token-b"); + assert.notEqual( + table.resolve("token-a").durableObjectId, + table.resolve("token-b").durableObjectId, + ); + }); +}); + +describe("Outbox — a pending row is never lost", () => { + const entry = (id: string) => ({ + id, + tenantId: "t_a", + payload: { n: 1 }, + createdAt: 100, + }); + + test("an enqueued entry is pending until settled", () => { + const box = new Outbox(); + box.enqueue(entry("e1")); + assert.equal(box.pending().length, 1); + assert.equal(box.settle("e1", 200), true); + assert.equal(box.pending().length, 0); + }); + + test("a second settle is a no-op, not a second publish", () => { + const box = new Outbox(); + box.enqueue(entry("e1")); + box.settle("e1", 200); + assert.equal( + box.settle("e1", 300), + false, + "retrying a settle must not republish", + ); + assert.equal(box.size, 1); + }); + + test("re-enqueuing the same id does not duplicate or clobber", () => { + const box = new Outbox(); + box.enqueue(entry("e1")); + box.enqueue({ ...entry("e1"), payload: { n: 999 } }); + assert.equal(box.size, 1); + const kept = box.pending()[0]; + assert.deepEqual( + kept?.payload, + { n: 1 }, + "the original payload is not overwritten", + ); + }); + + test("a crash before settle leaves the entry on the retry list", () => { + // The whole point: redelivery, not loss. + const box = new Outbox(); + box.enqueue(entry("e1")); + assert.equal( + box.pending().length, + 1, + "an unsettled entry survives for retry", + ); + }); + + test("settling an unknown entry reports false rather than throwing", () => { + assert.equal(new Outbox().settle("nope", 1), false); + }); +}); + +describe("CursorMerger — at-least-once, idempotent", () => { + const batch = (from: number, to: number) => + Array.from({ length: to - from + 1 }, (_, i) => ({ seq: from + i })); + + test("a first batch is fully accepted", () => { + const m = new CursorMerger(); + assert.equal(m.merge("s1", batch(0, 4)), 5); + assert.equal(m.afterSeq("s1"), 4); + }); + + test("a replayed batch changes nothing", () => { + // Deliberately NOT exactly-once — PROTOCOL.md reserves that for approvals. This is the + // at-least-once with an idempotent consumer, which is what the stream path is. + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + assert.equal( + m.merge("s1", batch(0, 4)), + 0, + "a full replay accepts nothing new", + ); + assert.equal(m.afterSeq("s1"), 4); + }); + + test("an overlapping batch accepts only the new tail", () => { + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + assert.equal( + m.merge("s1", batch(3, 8)), + 4, + "seq 3 and 4 were already held", + ); + assert.equal(m.afterSeq("s1"), 8); + }); + + test("out-of-order delivery is accepted correctly", () => { + // A host reconnecting may replay in any order; the cursor is by seq, not arrival. + const m = new CursorMerger(); + assert.equal(m.merge("s1", [{ seq: 3 }, { seq: 1 }, { seq: 2 }]), 3); + assert.equal( + m.afterSeq("s1"), + 3, + "the highest seq, regardless of arrival order", + ); + }); + + test("sessions are independent", () => { + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + assert.equal(m.merge("s2", batch(0, 2)), 3, "s2 has its own cursor"); + assert.equal(m.afterSeq("s1"), 4); + }); + + test("an empty batch is a no-op", () => { + const m = new CursorMerger(); + assert.equal(m.merge("s1", []), 0); + assert.equal(m.afterSeq("s1"), -1); + }); + + test("nextCursor tells the host exactly where to resume", () => { + const m = new CursorMerger(); + m.merge("s1", batch(0, 9)); + assert.deepEqual(m.nextCursor("s1"), { sessionId: "s1", afterSeq: 9 }); + }); + + test("resuming from a cursor delivers the remainder with no gap", () => { + // The offline-host loop: sync, go away, come back, resume. + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + const { afterSeq } = m.nextCursor("s1"); + const resumed = batch(afterSeq + 1, 9); + assert.equal( + m.merge("s1", resumed), + 5, + "exactly the records after the cursor", + ); + assert.equal(m.afterSeq("s1"), 9); + }); +}); From f5eff3386dab5a079584d3f658575ce6237a865e Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 12:08:37 +0530 Subject: [PATCH 6/7] docs,scripts: measure the runtime matrix, and generalise I15 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ADAPTERS.md opened with "verify, don't assume — this table will rot". This makes that a measurement rather than a recollection: `pnpm probe:runtimes` probes each installed CLI and emits the matrix, with `--json` for the machine-readable form. It is not in CI, because which CLIs are installed is an environment fact — a missing runtime is reported, not failed. The probe settles two things the docs previously only asserted. 1. "Sessions run YOLO by default" is now VERIFIED, not quoted. Four of five CLIs expose a permission-bypass flag and oar passes it: claude --dangerously-skip-permissions, grok --always-approve, kimi --yolo. Only codex documents a sandbox (-s/--sandbox). The premise this entire product is built on is checkable, and it checks out. 2. FOUR OF FIVE RUNTIMES SELF-UPDATE — claude/codex/grok `update`, kimi `upgrade`. `k.managed-copy-never-self-upgrades` (I15) was recorded for k-carrier, but the reasoning is not specific to it: a managed copy that also upgrades itself is a second writer on a component the host believes it owns. If Radius ever manages a copy of any of these four — exactly the k-carrier pattern — their built-in updater is that same second writer. So I15's scope is every managed binary on the host, not one vendored component, and this probe is how the condition is noticed appearing somewhere new. Still unmeasured, and stated as such in the doc and SAFETY.md §3: a documented flag is not a working sandbox. This measures what each CLI CLAIMS to support; the isolation it actually enforces is unknown. Writing the probe caught a stray TypeScript `as` in a .mjs file — caught by pnpm lint, which is the gate doing its job. Signed-off-by: Lakshman Patel --- docs/ADAPTERS.md | 35 +++++++++ package.json | 1 + scripts/probe-runtimes.mjs | 143 +++++++++++++++++++++++++++++++++++++ 3 files changed, 179 insertions(+) create mode 100644 scripts/probe-runtimes.mjs diff --git a/docs/ADAPTERS.md b/docs/ADAPTERS.md index 3d2b334..d5d19e5 100644 --- a/docs/ADAPTERS.md +++ b/docs/ADAPTERS.md @@ -49,6 +49,41 @@ for a human at a terminal. Wrong for unattended agents. We are the counter-case, the opposite position — and should document that we disagree with our own dependency, on the record. +## 2. MEASURED, not assumed — re-run with `pnpm probe:runtimes` + +Probed 2026-09-28 against the installed CLIs. This is a **measurement**, not a recollection, and +it is re-runnable so the table cannot silently rot. + +| Runtime | Version | Documents a sandbox | Permission-bypass flag | Self-updates | +| ------- | --------------------- | ------------------- | ---------------------------- | ------------------- | +| claude | 2.1.283 (Claude Code) | no | **yes** | **yes** (`update`) | +| codex | codex-cli 0.157.1 | **yes** (`-s`) | — | **yes** (`update`) | +| grok | grok 1.0.25 | no | **yes** (`--always-approve`) | **yes** (`update`) | +| kimi | 0.26.0 | no | **yes** (`--yolo`) | **yes** (`upgrade`) | +| pi | 0.85.1 | no | — | — | + +**Two things this measurement settles that the prose above only asserted.** + +**1. "Sessions run YOLO by default" is now a verified fact, not a quote.** Four of five CLIs +expose a permission-bypass flag, and oar passes it (`claude --dangerously-skip-permissions`, +`grok --always-approve`, `kimi --yolo`). The premise this whole product is built on is +checkable, and it checks out. Only Codex has a documented sandbox at all. + +**2. FOUR OF FIVE RUNTIMES SELF-UPDATE — and that generalises I15.** + +`k.managed-copy-never-self-upgrades` was recorded for k-carrier: a copy we install and manage +must never also upgrade itself, or there are two writers on one component. That reasoning is not +specific to k-carrier. If Radius ever installs and manages a copy of any of the four CLIs above — +exactly the k-carrier pattern — their built-in `update`/`upgrade` is **the same second writer**. + +So the invariant's scope is wider than one vendored component: **every managed binary on the +host must be upgradeable only by the supervisor.** Recorded in `THREATS.md` as I15, with the +vendor check asserting k-carrier's copy; this probe is how we notice the condition appearing in a +new place. + +**A documented flag is not a working sandbox.** This measures what each CLI _claims_ to support. +The isolation any of them actually enforces is still unmeasured, and `SAFETY.md` §3 says so. + --- ## 3. Capabilities oar exposes diff --git a/package.json b/package.json index bb7cea8..3c69abb 100644 --- a/package.json +++ b/package.json @@ -32,6 +32,7 @@ "check:invariants": "node scripts/check-invariants.mjs", "check:notices": "node scripts/check-notices.mjs", "check:proofs": "node scripts/check-proofs.mjs", + "probe:runtimes": "node scripts/probe-runtimes.mjs", "clean": "turbo run clean && rm -rf node_modules", "prepare": "husky" }, diff --git a/scripts/probe-runtimes.mjs b/scripts/probe-runtimes.mjs new file mode 100644 index 0000000..c4be17f --- /dev/null +++ b/scripts/probe-runtimes.mjs @@ -0,0 +1,143 @@ +#!/usr/bin/env node +/** + * Probe the installed agent CLIs and emit a *measured* capability matrix. + * + * `ADAPTERS.md` opens with "verify, don't assume — this table will rot". This script is the + * verification, so the table is a measurement rather than a recollection. It is not in CI: it + * depends on which CLIs happen to be installed, so a missing runtime is reported, not failed. + * + * Why it matters more than a version list. Two claims this project rests on are checkable here: + * + * 1. **"Sessions run YOLO by default."** That is the premise of the entire product. It is + * checkable: if each CLI exposes a permission-bypass flag, and oar passes it, then the + * premise is a fact rather than a quotation from a contract file. + * 2. **The managed-copy invariant (I15) is not specific to k-carrier.** Four of the five + * runtimes ship their own `update`/`upgrade` subcommand. If Radius ever installs and + * manages a copy of one of them — exactly the k-carrier pattern — that self-update is a + * second writer on a component the host believes it owns. `k.managed-copy-never-self-upgrades` + * therefore generalises from our installer to every managed binary, and this probe is how + * we notice when a new one appears. + * + * Run: pnpm probe:runtimes (human-readable) + * pnpm probe:runtimes -- --json (machine-readable, for the matrix check) + */ +import { execFileSync } from "node:child_process"; +import { writeFileSync } from "node:fs"; +import { join } from "node:path"; + +const ROOT = new URL("..", import.meta.url).pathname; +const AS_JSON = process.argv.includes("--json"); + +/** The five runtimes oar ships adapters for (`@botiverse/oar` index.d.ts). */ +const RUNTIMES = ["claude", "codex", "grok", "kimi", "pi"]; + +/** + * Patterns are matched against `--help` output. They are deliberately conservative: a match + * means the CLI *documents* the capability, not that we have verified its enforcement. The + * `documented` prefix in the field names is the honesty — a flag existing is not a sandbox + * working, and `docs/SAFETY.md` §3 is where that distinction is recorded. + */ +const PROBES = { + version: /^\s*(\d+\.\d+\.\d+)/m, + nativeSandbox: + /\b(--sandbox|-s, --sandbox|sandbox mode|workspace-write|danger-full-access)\b/i, + permissionBypass: + /(--dangerously-skip-permissions|--allow-dangerously-skip-permissions|--always-approve|--yolo\b|always-approve)/i, + sandboxNetworkNote: /no internet access/i, + selfUpdate: /^\s+(update|upgrade)\b/m, + workspaceDir: /(--add-dir|workspace directory|working directory)/i, + toolAllowlist: + /(--allowedTools|--disallowedTools|permission (allow|deny) rule|--tools\b)/i, +}; + +function probe(runtime) { + let help = ""; + let installed = true; + let version = null; + try { + help = execFileSync(runtime, ["--help"], { + encoding: "utf8", + timeout: 25_000, + }); + } catch { + installed = false; + } + + if (installed) { + try { + version = execFileSync(runtime, ["--version"], { + encoding: "utf8", + timeout: 25_000, + }) + .toString() + .trim(); + } catch { + version = null; + } + } + + const flags = Object.fromEntries( + Object.entries(PROBES) + .filter(([k]) => k !== "version" && k !== "selfUpdate") + .map(([k, re]) => [k, installed && re.test(help)]), + ); + + return { + runtime, + installed, + version, + // Documented, not enforced. Nothing here proves a sandbox works. + ...flags, + selfUpdates: installed && PROBES.selfUpdate.test(help), + updateCommand: help.match(/^\s+(update|upgrade)\b/m)?.[1] ?? null, + }; +} + +const results = RUNTIMES.map(probe); + +if (AS_JSON) { + writeFileSync( + join(ROOT, "docs/runtimes.probe.json"), + `${JSON.stringify(results, null, 2)}\n`, + ); + console.log(`wrote docs/runtimes.probe.json (${results.length} runtimes)`); + process.exit(0); +} + +const yesNo = (b) => (b === null ? "n/a" : b ? "YES" : "-"); +const pad = (s, n) => String(s ?? "-").padEnd(n); + +console.log( + `\nRuntime capability probe — ${new Date().toISOString().slice(0, 10)}\n`, +); +console.log( + ` ${pad("runtime", 8)}${pad("version", 26)}${pad("sandbox?", 10)}${pad("perm-bypass", 12)}self-update`, +); +console.log(` ${"-".repeat(74)}`); +for (const r of results) { + console.log( + ` ${pad(r.runtime, 8)}${pad(r.version, 26)}${pad(yesNo(r.nativeSandbox), 10)}` + + `${pad(yesNo(r.permissionBypass), 12)}${yesNo(r.selfUpdates)}${ + r.selfUpdates ? ` (${r.updateCommand})` : "" + }`, + ); +} + +const missing = results.filter((r) => !r.installed); +if (missing.length > 0) { + console.log( + `\n not installed: ${missing.map((m) => m.runtime).join(", ")} — reported, not failed`, + ); +} + +const selfUpdating = results.filter((r) => r.selfUpdates); +console.log( + `\n ${selfUpdating.length} of ${results.length} runtimes ship a self-update path. If Radius ` + + `ever manages a copy of one, that is a SECOND WRITER on a component the host believes it ` + + `owns — the same shape as k.managed-copy-never-self-upgrades (I15), which therefore ` + + `generalises beyond k-carrier.`, +); +console.log( + `\n A documented flag is not a working sandbox. This measures what each CLI CLAIMS to support;\n` + + ` the isolation it actually enforces is still unmeasured — see SAFETY.md §3.\n`, +); From 64b15fc2804feb8c367013a48eb8f97b7cacee4d Mon Sep 17 00:00:00 2001 From: Lakshman Patel Date: Mon, 28 Sep 2026 13:53:16 +0530 Subject: [PATCH 7/7] feat(protocol): radius/v1 versioning, enforced rather than promised MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ROADMAP calls the public API "the real moat", which only holds if versioning is enforced instead of described. PROTOCOL.md §8's rules are implemented here as a checker: parse, classify, and assert the bump. The rule that matters is the one that surprises people: TIGHTENING A DEFAULT IS A MAJOR. Turning a sandbox on by default is breaking for a host built against v1 defaults even though no field changed — a host that silently loses its sandbox must fail loudly, not drift. A tool that diffs field names cannot see that, so classifyRelease works on semantic facts, not shapes. assertVersionBump fails in BOTH directions, which is the part most teams get wrong. Direction one (a breaking change shipped as a minor) breaks your own client, so everyone catches it. Direction two (an additive change shipped as a major) never breaks anyone — it just forces an upgrade for an optional field, until "stable" means nothing. It feels safe precisely because it is quiet. assertCompatible implements "a host may lag the plane by one minor, never a major", including refusing a two-minor lag: "one" means one, and assuming two is how a compatibility promise quietly dies. MUTATION-VERIFIED (5/5) a tightened default downgraded to minor (3 tests fail), a two-minor lag accepted, a major mismatch accepted, an additive change allowed to bump major, and version parsing made permissive. The first attempt at the first mutation left a dangling `||` and produced a syntax error rather than a real failure — a false catch, redone properly. A mutation that fails for the wrong reason has verified nothing. STILL NOT BUILT: the wire transport, an SDK, and the compatibility policy as a published document. This is the version surface those would sit on. 167 tests, 8/8 gates. Signed-off-by: Lakshman Patel --- packages/protocol/src/index.ts | 19 +++ packages/protocol/src/version.ts | 177 ++++++++++++++++++++ packages/protocol/test/version.test.ts | 218 +++++++++++++++++++++++++ 3 files changed, 414 insertions(+) create mode 100644 packages/protocol/src/version.ts create mode 100644 packages/protocol/test/version.test.ts diff --git a/packages/protocol/src/index.ts b/packages/protocol/src/index.ts index 27501d1..a50d7c5 100644 --- a/packages/protocol/src/index.ts +++ b/packages/protocol/src/index.ts @@ -63,3 +63,22 @@ export { StaticTenantTable, UnresolvedTenantError, } from "./sync.ts"; + +export type { + ChangeClass, + Envelope, + ParsedVersion, + ReleaseNote, +} from "./version.ts"; +export { + PROTOCOL_MAJOR, + PROTOCOL_MINOR, + PROTOCOL_VERSION, + BreakingReleaseError, + UnsupportedVersionError, + assertCompatible, + assertVersionBump, + classifyRelease, + formatVersion, + parseVersion, +} from "./version.ts"; diff --git a/packages/protocol/src/version.ts b/packages/protocol/src/version.ts new file mode 100644 index 0000000..32b36b4 --- /dev/null +++ b/packages/protocol/src/version.ts @@ -0,0 +1,177 @@ +/** + * `radius/v1` — the envelope, and the compatibility rules that make a promise you can keep. + * + * `ROADMAP.md` calls this "the real moat": once developers build agents against `radius/v1`, the + * platform gets stickier than any feature set. That only holds if versioning is *enforced* rather + * than promised, so the rules from `PROTOCOL.md` §8 are implemented here as a checker rather than + * left as prose. + * + * ── The rules, and the one that surprises people ─────────────────────────────────────────── + * new optional field ............ minor + * new record kind ............... minor + * removal, rename, semantic ..... MAJOR + * **tightening a default ....... MAJOR** + * + * That last one is load-bearing and the reason this file exists. Turning a sandbox on by default + * is a *breaking* change for a host built against v1 defaults, even though no field changed — a + * host that silently loses its sandbox must fail loudly, not drift. A tool that only diffs field + * names cannot see that, so `classifyRelease` works on **semantic** facts, not shapes. + * + * "A host may lag the plane by one minor. Never a major." — enforced in `assertCompatible`. + */ + +import { ProtocolError } from "./principal.ts"; + +// Re-exported so a consumer of the version surface catches the malformed-version case without +// also reaching into the principal module. Same as `lease.ts`. +export { ProtocolError }; + +export const PROTOCOL_VERSION = "radius/v1"; +export const PROTOCOL_MAJOR = 1; +export const PROTOCOL_MINOR = 0; + +export interface ParsedVersion { + readonly major: number; + readonly minor: number; +} + +/** Parse `radius/vN` or `radius/vN.M`. Rejects anything else rather than guessing. */ +export function parseVersion(raw: string): ParsedVersion { + const match = /^radius\/v(\d+)(?:\.(\d+))?$/.exec(raw); + if (!match?.[1]) { + throw new ProtocolError( + `unrecognised protocol version "${raw}". Expected ${PROTOCOL_VERSION} or radius/vN.M.`, + ); + } + return { major: Number(match[1]), minor: Number(match[2] ?? 0) }; +} + +export function formatVersion({ major, minor }: ParsedVersion): string { + return `radius/v${major}.${minor}`; +} + +/** The wire envelope. `v` is required — `PROTOCOL.md` §2 says so explicitly. */ +export interface Envelope { + readonly v: string; + readonly requestId: string; + readonly payload: T; +} + +export class UnsupportedVersionError extends Error { + constructor( + readonly theirs: string, + readonly ours: string, + ) { + super(`unsupported protocol version ${theirs}; this build speaks ${ours}`); + this.name = "UnsupportedVersionError"; + } +} + +/** + * Can a host at `host` talk to a plane at `plane`? + * + * `PROTOCOL.md` §8: "A host may lag the plane by one minor. Never a major." So a newer *minor* on + * the host is tolerated, a major mismatch is not, and more than one minor of lag is not. Being + * strict here is the point — a silently-mismatched pair is how a "backwards compatible" API + * stops being one. + */ +export function assertCompatible(plane: string, host: string): void { + const p = parseVersion(plane); + const h = parseVersion(host); + if (p.major !== h.major) { + throw new UnsupportedVersionError(host, plane); + } + if (p.minor - h.minor > 1) { + throw new ProtocolError( + `host is ${p.minor - h.minor} minors behind the plane, and only one minor of lag is ` + + `supported. Upgrade the host rather than assuming compatibility.`, + ); + } +} + +/** A semantic fact about a release — deliberately not a field-level diff. */ +export interface ReleaseNote { + readonly version: string; + readonly newOptionalFields?: readonly string[]; + readonly newRecordKinds?: readonly string[]; + readonly removedOrRenamed?: readonly string[]; + readonly semanticChanges?: readonly string[]; + /** A default that got STRICTER, e.g. sandbox on. This is a major, by rule. */ + readonly tightenedDefaults?: readonly string[]; +} + +export type ChangeClass = "minor" | "major" | "none"; + +/** + * Classify one release. Pure, so the policy is tested rather than trusted. + * + * The two interesting outputs are the ones that are easy to get wrong: a tightened default is + * **major** even with nothing added, and an *empty* release is neither minor nor major. + */ +export function classifyRelease(note: ReleaseNote): ChangeClass { + if ( + (note.removedOrRenamed?.length ?? 0) > 0 || + (note.semanticChanges?.length ?? 0) > 0 || + (note.tightenedDefaults?.length ?? 0) > 0 + ) { + return "major"; + } + if ( + (note.newOptionalFields?.length ?? 0) > 0 || + (note.newRecordKinds?.length ?? 0) > 0 + ) { + return "minor"; + } + return "none"; +} + +export class BreakingReleaseError extends Error { + constructor( + readonly version: string, + readonly reasons: readonly string[], + ) { + super( + `release ${version} is breaking: ${reasons.join("; ")}. A tightened default counts as ` + + `breaking even when no field changed — a host that silently loses its sandbox must fail ` + + `loudly, not drift. Bump the major version.`, + ); + this.name = "BreakingReleaseError"; + } +} + +/** + * Assert a release does the right thing with its version number. + * + * Fails in both directions, deliberately: a breaking change shipped as a minor, *and* a minor + * shipped as a major. The second looks harmless and is not — it forces every host to upgrade for + * an additive change, which is exactly how a "stable API" becomes a moving target. + */ +export function assertVersionBump(previous: string, note: ReleaseNote): void { + const from = parseVersion(previous); + const to = parseVersion(note.version); + const kind = classifyRelease(note); + + if (to.major < from.major) { + throw new ProtocolError( + `version went backwards: ${previous} → ${note.version}`, + ); + } + + if (kind === "major" && to.major === from.major) { + throw new BreakingReleaseError( + note.version, + [ + ...(note.removedOrRenamed ?? []).map((f) => `removed/renamed ${f}`), + ...(note.semanticChanges ?? []).map((c) => `semantic: ${c}`), + ...(note.tightenedDefaults ?? []).map((d) => `tightened default: ${d}`), + ].slice(0, 3), + ); + } + + if (kind === "minor" && to.major > from.major) { + throw new ProtocolError( + `${note.version} is an additive change but bumped the MAJOR version. That forces every ` + + `host to upgrade for an optional field; it is a minor.`, + ); + } +} diff --git a/packages/protocol/test/version.test.ts b/packages/protocol/test/version.test.ts new file mode 100644 index 0000000..0828357 --- /dev/null +++ b/packages/protocol/test/version.test.ts @@ -0,0 +1,218 @@ +/** + * `radius/v1` versioning and compatibility. + * + * The tests that matter most are the two that fail in *both* directions — a breaking change + * shipped as a minor, and a minor shipped as a major. The second is the one a careful team gets + * wrong, because it feels safe: it only ever inconveniences people, never breaks them. It is + * also how a "stable API" quietly becomes a moving target. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + PROTOCOL_VERSION, + BreakingReleaseError, + ProtocolError, + UnsupportedVersionError, + assertCompatible, + assertVersionBump, + classifyRelease, + formatVersion, + parseVersion, +} from "../src/version.ts"; + +describe("version parsing", () => { + test("parses the protocol version and an explicit minor", () => { + assert.deepEqual(parseVersion(PROTOCOL_VERSION), { major: 1, minor: 0 }); + assert.deepEqual(parseVersion("radius/v2.7"), { major: 2, minor: 7 }); + }); + + test("round-trips through format", () => { + assert.equal(formatVersion(parseVersion("radius/v3.4")), "radius/v3.4"); + }); + + test("rejects anything unrecognised rather than guessing", () => { + for (const bad of [ + "v1", + "radius/1", + "radius", + "latest", + "", + "radius/v1.2.3", + ]) { + assert.throws( + () => parseVersion(bad), + ProtocolError, + `"${bad}" must be rejected`, + ); + } + }); +}); + +describe("compatibility — a host may lag one minor, never a major", () => { + test("the same version, and a host one minor behind, are compatible", () => { + assert.doesNotThrow(() => assertCompatible("radius/v1.2", "radius/v1.2")); + assert.doesNotThrow(() => assertCompatible("radius/v1.2", "radius/v1.1")); + }); + + test("a host AHEAD on the minor is compatible", () => { + // A newer host talking to an older plane is fine for additive changes. + assert.doesNotThrow(() => assertCompatible("radius/v1.1", "radius/v1.2")); + }); + + test("a host two minors behind is NOT compatible", () => { + // "Only one minor" means one. Assuming two is how a compat promise quietly dies. + assert.throws( + () => assertCompatible("radius/v1.3", "radius/v1.1"), + /minors behind/, + ); + }); + + test("a major mismatch is refused in both directions", () => { + assert.throws( + () => assertCompatible("radius/v2.0", "radius/v1.9"), + UnsupportedVersionError, + ); + assert.throws( + () => assertCompatible("radius/v1.9", "radius/v2.0"), + UnsupportedVersionError, + ); + }); + + test("a malformed version is refused, not coerced", () => { + assert.throws( + () => assertCompatible("radius/v1.0", "garbage"), + ProtocolError, + ); + }); +}); + +describe("change classification", () => { + test("a new optional field or record kind is a minor", () => { + assert.equal( + classifyRelease({ version: "radius/v1.1", newOptionalFields: ["scope"] }), + "minor", + ); + assert.equal( + classifyRelease({ version: "radius/v1.1", newRecordKinds: ["handoff"] }), + "minor", + ); + }); + + test("a removal, rename, or semantic change is a MAJOR", () => { + assert.equal( + classifyRelease({ version: "radius/v2.0", removedOrRenamed: ["scope"] }), + "major", + ); + assert.equal( + classifyRelease({ + version: "radius/v2.0", + semanticChanges: ["seq now starts at 1"], + }), + "major", + ); + }); + + test("a TIGHTENED DEFAULT is a MAJOR, with no field changed at all", () => { + // The rule that surprises people, and the reason classifyRelease works on semantics rather + // than shapes. Turning a sandbox on cannot be additive: a host built against v1 defaults must + // fail loudly, not discover it later. + assert.equal( + classifyRelease({ + version: "radius/v2.0", + tightenedDefaults: ["sandbox on by default"], + }), + "major", + ); + }); + + test("an empty release is neither minor nor major", () => { + assert.equal(classifyRelease({ version: "radius/v1.0" }), "none"); + }); +}); + +describe("version bump enforcement", () => { + test("a minor additive change passes", () => { + assert.doesNotThrow(() => + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + newOptionalFields: ["scope"], + }), + ); + }); + + test("a major breaking change passes", () => { + assert.doesNotThrow(() => + assertVersionBump("radius/v1.0", { + version: "radius/v2.0", + tightenedDefaults: ["sandbox on"], + }), + ); + }); + + test("a breaking change shipped as a minor is REFUSED", () => { + // Direction one. The mistake everyone catches, because their own client breaks. + assert.throws( + () => + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + semanticChanges: ["cursor semantics changed"], + }), + BreakingReleaseError, + ); + }); + + test("a tightened default shipped as a minor is REFUSED", () => { + assert.throws( + () => + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + tightenedDefaults: ["sandbox on by default"], + }), + /breaking/, + ); + }); + + test("an additive change shipped as a MAJOR is REFUSED", () => { + // Direction two, and the subtler one. It never breaks anyone; it just forces an upgrade for + // an optional field, until "stable" means nothing. + assert.throws( + () => + assertVersionBump("radius/v1.0", { + version: "radius/v2.0", + newOptionalFields: ["scope"], + }), + /MAJOR version/, + ); + }); + + test("a version that goes backwards is refused", () => { + assert.throws( + () => + assertVersionBump("radius/v2.0", { + version: "radius/v1.9", + newOptionalFields: ["x"], + }), + /backwards/, + ); + }); + + test("the breaking error names what broke", () => { + // "This is breaking" is not actionable; naming the field is. + try { + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + removedOrRenamed: ["scope"], + semanticChanges: ["seq base changed"], + }); + assert.fail("should have thrown"); + } catch (e) { + assert.ok(e instanceof BreakingReleaseError); + assert.ok( + e.reasons.some((r) => r.includes("scope")), + "names the removed field", + ); + } + }); +});