diff --git a/apps/agent-host/src/supervisor.ts b/apps/agent-host/src/supervisor.ts new file mode 100644 index 0000000..e3b2f5b --- /dev/null +++ b/apps/agent-host/src/supervisor.ts @@ -0,0 +1,147 @@ +import type { RuntimeRegistry } from "@botiverse/oar"; + +import type { RecordStore } from "./record-store.ts"; +import { startHostSession, type HostSession } from "./session-host.ts"; + +export type SupervisorState = "idle" | "running" | "stopping"; + +/** `start()` was called while a session was already running. */ +export class AlreadyRunningError extends Error { + constructor(readonly sessionId: string) { + super( + `a session for "${sessionId}" is already running. Two writers on one stream is the ` + + `split-brain the vendored proofs are about — use restart() to cycle deliberately.`, + ); + this.name = "AlreadyRunningError"; + } +} + +/** `stop()` did not finish inside the budget, so the process may still be resident. */ +export class StopTimeoutError extends Error { + constructor( + readonly sessionId: string, + readonly timeoutMs: number, + ) { + super( + `stopping "${sessionId}" did not complete within ${timeoutMs}ms. The harness may still be ` + + `running — this is reported rather than swallowed, because "it probably exited" is not a ` + + `state we are willing to claim.`, + ); + this.name = "StopTimeoutError"; + } +} + +export interface AgentSupervisorOptions { + readonly runtimeId: string; + readonly cwd: string; + readonly store: RecordStore; + /** Durable stream name. See `StartHostSessionOptions.sessionId` — one stream per Session. */ + readonly sessionId: string; + readonly model?: string; + readonly resume?: string; + readonly registry?: RuntimeRegistry; + /** + * Budget for `stop()`. An unattended host cannot hang forever on a wedged harness, and a + * silent hang is indistinguishable from a leak. Exceeding it raises `StopTimeoutError`. + */ + readonly stopTimeoutMs?: number; +} + +const DEFAULT_STOP_TIMEOUT_MS = 30_000; + +/** + * Owns the lifecycle of exactly one agent session. + * + * Small on purpose. The valuable part is not the start/stop plumbing — it is the three + * properties this class exists to make true, each of which is a way unattended agents go wrong: + * + * 1. **Never two writers.** `start()` refuses while running. Two drivers on one stream is the + * same split-brain k-carrier's `never_dual_run` is about, one layer up. `restart()` is the + * only way to cycle, so the intent is explicit. + * 2. **Nothing observed is left unpersisted.** `stop()` drains before it returns, even when + * `dispose()` throws. The exit records are the ones an operator wants most. + * 3. **No process residency.** `stop()` does not resolve until the harness is gone — and if it + * cannot prove that within the budget, it says so instead of returning quietly. + * + * State transitions are one-way and total: idle → running → idle, with `stopping` observable + * while a stop is in flight. There is no state from which a second session can appear. + */ +export class AgentSupervisor { + readonly #options: AgentSupervisorOptions; + #state: SupervisorState = "idle"; + #current: HostSession | null = null; + + constructor(options: AgentSupervisorOptions) { + this.#options = options; + } + + get state(): SupervisorState { + return this.#state; + } + + /** The live session, or null. Exposed for the caller to prompt/steer; not for lifecycle. */ + get session(): HostSession | null { + return this.#current; + } + + get sessionId(): string { + return this.#options.sessionId; + } + + async start(): Promise { + if (this.#state !== "idle") { + throw new AlreadyRunningError(this.#options.sessionId); + } + // A new stream per Session, so a restart gets a fresh name unless the caller is resuming a + // brand-new stream into the same file. See StartHostSessionOptions.sessionId — reusing a + // name across a fresh seq-0 stream is a lower-seq append, and RecordStore rejects it loudly. + const host = await startHostSession({ + runtimeId: this.#options.runtimeId, + cwd: this.#options.cwd, + store: this.#options.store, + sessionId: this.#options.sessionId, + model: this.#options.model, + resume: this.#options.resume, + registry: this.#options.registry, + }); + this.#current = host; + this.#state = "running"; + return host; + } + + /** + * Stop the session and leave nothing behind. Idempotent. + * + * The timeout is a real ceiling, not a formality: a wedged harness must not pin the host + * forever. It raises rather than resolves quietly, because a supervisor that returns "stopped" + * while a process is still running is precisely the lie an unattended deployment cannot catch. + */ + async stop(): Promise { + const current = this.#current; + if (!current || this.#state === "stopping") return; + this.#state = "stopping"; + + const budget = this.#options.stopTimeoutMs ?? DEFAULT_STOP_TIMEOUT_MS; + let timer: ReturnType | undefined; + const timeout = new Promise((_, reject) => { + timer = setTimeout( + () => reject(new StopTimeoutError(this.#options.sessionId, budget)), + budget, + ); + }); + + try { + await Promise.race([current.stop(), timeout]); + } finally { + if (timer) clearTimeout(timer); + this.#current = null; + this.#state = "idle"; + } + } + + /** Stop, then start. The only supported way to cycle a session. */ + async restart(): Promise { + await this.stop(); + return this.start(); + } +} diff --git a/apps/agent-host/test/durability-harness.test.ts b/apps/agent-host/test/durability-harness.test.ts new file mode 100644 index 0000000..12485d1 --- /dev/null +++ b/apps/agent-host/test/durability-harness.test.ts @@ -0,0 +1,144 @@ +/** + * Milestone 0.4 — the stream durability harness. + * + * `MILESTONES.md` 0.4 asks for two things, and this file exists to make both true: + * + * 1. "Fuzz the kill point across the upgrade state machine." + * 2. "Assert `never_dual_run` and `never_bricked` behaviourally, not just by trusting the proofs." + * + * The Lean proofs are about k-carrier's two-slot upgrade, and we verified them separately + * (`pnpm check:proofs`). They say nothing about OUR record path. This harness is the equivalent + * for the stream: it interrupts a write at every point where interruption is possible, and then + * asserts what survived rather than what should have. + * + * ── What is actually asserted ────────────────────────────────────────────────────────────── + * Not "nothing was lost" — that is false, and the RecordWriter documents why: a record observed + * but not yet fsync'd dies with the process, and no amount of testing makes that not true. + * + * Instead, the two properties that can be absolute, and which together are what an operator + * actually needs after a hard kill: + * + * - **The surviving stream is a valid prefix.** Never a gap, never a torn record, never a + * record whose body is half-written. Losing the tail is acceptable; a corrupt prefix is not. + * - **The stream is monotonic and unique.** No `seq` appears twice. A duplicate is worse than a + * loss, because a consumer would double-apply it. + * + * That is the contract the control plane's cursor sync depends on. If it holds after a kill at + * any point, the plane can resume from `lastSeq` and never sees a lie. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +import { RecordStore } from "../src/record-store.ts"; + +describe("durability harness — a kill at any point leaves a valid prefix", () => { + test("a torn final line never corrupts the records before it", async () => { + // The realistic hard-kill artefact: a partial write at the tail. Every complete line before + // it must still parse and still be in order. + for (let killAt = 1; killAt <= 40; killAt++) { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-fuzz-")); + const store = new RecordStore({ dir }); + const complete = 5; + for (let i = 0; i < complete; i++) { + await store.append("s", { + seq: i, + sessionId: "s", + kind: "frame", + body: { n: i }, + }); + } + // Simulate a kill mid-write by appending a partial line of `killAt` bytes. + fs.appendFileSync( + path.join(dir, "s.jsonl"), + '{"seq":99,"b'.slice(0, killAt % 12), + ); + await store.close(); + + const seqs: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s", -1)) + seqs.push(r.seq); + assert.deepEqual( + seqs, + Array.from({ length: complete }, (_, i) => i), + `torn tail of ${killAt % 12} bytes must not disturb the ${complete} complete records`, + ); + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + + test("a stream is a valid prefix after truncation at every byte offset", async () => { + // Stronger than appending a torn line: take a real stream and truncate the FILE at every + // possible offset. Whatever remains must be a valid prefix — this catches a partial line in + // the middle of the file, which the tail-only test would miss. + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-trunc-")); + const store = new RecordStore({ dir }); + const total = 8; + for (let i = 0; i < total; i++) { + await store.append("s", { + seq: i, + sessionId: "s", + kind: "frame", + body: { n: i }, + }); + } + await store.close(); + + const file = path.join(dir, "s.jsonl"); + const full = fs.readFileSync(file, "utf8"); + for (let cut = 0; cut <= full.length; cut++) { + fs.writeFileSync(file, full.slice(0, cut)); + const seqs: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s", -1)) + seqs.push(r.seq); + // Must be exactly [0..k] for some k: a valid prefix, with no gaps and no duplicates. + assert.deepEqual( + seqs, + seqs.map((_, i) => i), + `truncating at byte ${cut} must leave a contiguous prefix, got ${seqs.join(",")}`, + ); + } + fs.rmSync(dir, { recursive: true, force: true }); + }); + + test("replaying a truncated stream resumes with no gap and no duplicate", async () => { + // The property the hybrid sync model actually depends on: a cursor taken from a surviving + // prefix, replayed against the full stream, is consistent. + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-resume-")); + const store = new RecordStore({ dir }); + const total = 10; + for (let i = 0; i < total; i++) { + await store.append("s", { + seq: i, + sessionId: "s", + kind: "frame", + body: { n: i }, + }); + } + await store.close(); + + const file = path.join(dir, "s.jsonl"); + const full = fs.readFileSync(file, "utf8"); + // Cut somewhere in the middle and take a cursor from what survived. + fs.writeFileSync(file, full.slice(0, Math.floor(full.length / 2))); + + const survivor = new RecordStore({ dir }); + let lastSeq = -1; + for await (const r of survivor.readAfter("s", -1)) lastSeq = r.seq; + + fs.writeFileSync(file, full); // the rest arrives after a reconnect + const resumed: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s", lastSeq)) + resumed.push(r.seq); + + assert.deepEqual( + resumed, + Array.from({ length: total - 1 - lastSeq }, (_, i) => lastSeq + 1 + i), + "resuming from a prefix cursor yields every later record exactly once", + ); + fs.rmSync(dir, { recursive: true, force: true }); + }); +}); diff --git a/apps/agent-host/test/e2e-kill.test.ts b/apps/agent-host/test/e2e-kill.test.ts index 9be22d5..320610d 100644 --- a/apps/agent-host/test/e2e-kill.test.ts +++ b/apps/agent-host/test/e2e-kill.test.ts @@ -44,6 +44,12 @@ import { RecordStore } from "../src/record-store.ts"; const ENABLED = process.env.RADIUS_E2E === "1"; const RUNTIME = process.env.RADIUS_E2E_RUNTIME ?? "claude"; +/** + * How long the child is allowed to run before it SIGKILLs itself. Short by default so the + * kill has a real chance of landing mid-generation rather than after a completed turn. + * Lower it further to aim at a colder harness. + */ +const KILL_MS = Number(process.env.RADIUS_E2E_KILL_MS ?? "12000"); const THIS_DIR = path.dirname(fileURLToPath(import.meta.url)); const SESSION_HOST = new URL("../src/session-host.ts", import.meta.url).href; @@ -77,7 +83,7 @@ describe("e2e — a real session killed mid-turn", { skip: SKIP }, () => { host.session.events(() => {}); process.stdout.write("READY\\n"); - setInterval(() => {}, 1000); + setTimeout(() => { process.kill(process.pid, "SIGKILL"); }, ${KILL_MS}); `; // spawnSync with a timeout delivers a REAL SIGKILL from the OS — not a simulated failure. @@ -100,16 +106,32 @@ describe("e2e — a real session killed mid-turn", { skip: SKIP }, () => { `stderr: ${child.stderr.slice(0, 800)}`, ); - const seqs: number[] = []; - for await (const r of new RecordStore({ dir }).readAfter(streamId, -1)) { - seqs.push(r.seq); - } + const records = []; + for await (const r of new RecordStore({ dir }).readAfter(streamId, -1)) + records.push(r); + const seqs = records.map((r) => r.seq); assert.ok( seqs.length > 0, "the killed session produced records; otherwise this proves nothing", ); + // Report which case this run actually hit, so a green tick cannot be read as proof of a + // mid-generation kill that may not have happened. A finished turn leaves a terminal + // `result/*` frame; its absence means output was still streaming when the kill landed. + const turnCompleted = records.some((r) => { + const native = (r.body as { native?: { type?: string } }).native; + return ( + typeof native?.type === "string" && native.type.startsWith("result/") + ); + }); + t.diagnostic( + `kill at ${KILL_MS}ms — ${seqs.length} records durable; turn ` + + (turnCompleted + ? "had COMPLETED before the kill" + : "was STILL IN FLIGHT at the kill"), + ); + // Contiguous from 0. A gap would mean a write vanished in a way the store cannot detect — // exactly the silent corruption this milestone exists to prevent. assert.deepEqual( diff --git a/apps/agent-host/test/supervisor.test.ts b/apps/agent-host/test/supervisor.test.ts new file mode 100644 index 0000000..3a4f40e --- /dev/null +++ b/apps/agent-host/test/supervisor.test.ts @@ -0,0 +1,345 @@ +/** + * Tests for `AgentSupervisor` — the lifecycle properties an unattended host depends on. + * + * The residency test is the important one, and it is deliberately NOT a mock assertion. It + * spawns a REAL child process, records its pid, stops the session, and then polls the OS until + * that pid is gone. A fake that merely records "dispose was called" would pass while the + * process lived on, which is the exact bug `agentNoProcessResidency` exists to prevent. + */ + +import assert from "node:assert/strict"; +import { test, describe, beforeEach, afterEach } from "node:test"; +import { spawn } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +import { + createRuntimeRegistry, + type InstallationProbe, + type InstallationSnapshot, + type RawEvent, + type RawEventObserver, + type Runtime, + type RuntimeRegistry, + type Session, + type Unsubscribe, +} from "@botiverse/oar"; + +import { RecordStore } from "../src/record-store.ts"; +import { + AgentSupervisor, + AlreadyRunningError, + StopTimeoutError, +} from "../src/supervisor.ts"; + +let dir: string; + +beforeEach(() => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), "radius-sup-")); +}); +afterEach(() => { + fs.rmSync(dir, { recursive: true, force: true }); +}); + +const AVAILABLE: InstallationSnapshot = { kind: "available", via: "bundled" }; +const probe: InstallationProbe = () => Promise.resolve(AVAILABLE); + +function frame(seq: number): RawEvent { + return { + kind: "frame", + seq, + sessionId: "native", + agentPath: [], + receivedAt: 1_700_000_000_000 + seq, + body: { type: "assistant", native: { text: `m${seq}` }, events: [] }, + }; +} + +/** A session whose dispose() actually kills a real OS process. */ +class ResidentSession implements Session { + readonly id = "native"; + readonly capabilities = { + steer: false, + queue: null, + attribution: "none", + } as Session["capabilities"]; + #observers = new Set(); + #seq = 0; + /** Set to make dispose() hang, to exercise the stop budget. */ + hangOnDispose = false; + + constructor( + readonly pid: number, + private readonly kill: (pid: number) => void, + ) {} + + rawEvents(observer: RawEventObserver): Unsubscribe { + this.#observers.add(observer); + return () => this.#observers.delete(observer); + } + emit(): void { + const r = frame(this.#seq++); + for (const o of this.#observers) o(r); + } + dispose(): Promise { + if (this.hangOnDispose) return new Promise(() => undefined); + this.emit(); // the exit record, as a real adapter produces + this.kill(this.pid); + this.#observers.clear(); + return Promise.resolve(); + } + records(): readonly RawEvent[] { + return []; + } + prompt(): Promise { + return Promise.reject(new Error("unused")); + } + steer(): Promise { + return Promise.reject(new Error("unused")); + } + queue(): Promise { + return Promise.reject(new Error("unused")); + } + abort(): Promise { + return Promise.reject(new Error("unused")); + } + graph() { + return { nodes: [], edges: [] }; + } + events(): Unsubscribe { + return () => undefined; + } + model() { + return { value: null, seq: -1 }; + } + usage() { + return { value: { total: null }, seq: -1 }; + } + contextUsage() { + return { value: null, seq: -1 }; + } + steerOrQueue(): Promise { + return Promise.reject(new Error("unused")); + } +} + +function registryFor(session: Session): RuntimeRegistry { + const runtime = { + id: "fake", + brand: { name: "Fake", icon: "data:image/svg+xml," }, + installation: probe, + session: () => Promise.resolve(session), + } as unknown as Runtime; + return createRuntimeRegistry([runtime]); +} + +describe("AgentSupervisor — lifecycle", () => { + test("start() refuses while a session is already running", async () => { + // Two writers on one stream is split-brain, the same failure the vendored proofs are + // about one layer up. restart() is the only supported way to cycle. + const store = new RecordStore({ dir }); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(new ResidentSession(0, () => undefined)), + }); + await sup.start(); + assert.equal(sup.state, "running"); + await assert.rejects(() => sup.start(), AlreadyRunningError); + await sup.stop(); + await store.close(); + }); + + test("stop() is idempotent, and stop() on an idle supervisor is a no-op", async () => { + const store = new RecordStore({ dir }); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(new ResidentSession(0, () => undefined)), + }); + await sup.stop(); // never started + assert.equal(sup.state, "idle"); + await sup.start(); + await sup.stop(); + await sup.stop(); + assert.equal(sup.state, "idle"); + assert.equal(sup.session, null); + await store.close(); + }); + + test("stop() persists the exit records before it returns", async () => { + // The records that say the agent shut down cleanly are the ones an operator wants most + // after a hard kill, so stop() must not resolve before they are durable. + const store = new RecordStore({ dir }); + const session = new ResidentSession(0, () => undefined); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + }); + await sup.start(); + session.emit(); + await sup.stop(); + await store.close(); + + const seqs: number[] = []; + for await (const r of new RecordStore({ dir }).readAfter("s1", -1)) + seqs.push(r.seq); + assert.deepEqual( + seqs, + [0, 1], + "the emitted frame and the exit record are both on disk", + ); + }); + + test("a throwing dispose still drains, and the supervisor returns to idle", async () => { + const store = new RecordStore({ dir }); + const session = new ResidentSession(0, () => undefined); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + }); + await sup.start(); + session.emit(); + // ResidentSession.dispose does not throw, so force the failure through the writer's + // sticky path instead: close the store underneath it. + await store.close(); + await sup.stop(); + assert.equal( + sup.state, + "idle", + "state recovers even when the stop path fails", + ); + assert.equal(sup.session, null); + }); + + test("a hung dispose raises StopTimeoutError instead of hanging forever", async () => { + // An unattended host cannot pin on a wedged harness, and a silent hang is + // indistinguishable from a leak. It must fail loudly. + const store = new RecordStore({ dir }); + const session = new ResidentSession(0, () => undefined); + session.hangOnDispose = true; + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + stopTimeoutMs: 120, + }); + await sup.start(); + await assert.rejects(() => sup.stop(), StopTimeoutError); + assert.equal( + sup.state, + "idle", + "a timed-out stop still releases the supervisor", + ); + await store.close(); + }); + + test("restart() stops first, then starts", async () => { + const store = new RecordStore({ dir }); + const a = new ResidentSession(0, () => undefined); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(a), + }); + await sup.start(); + // A restart replays seq from 0, so it MUST target a fresh stream. Reusing "s1" would be a + // lower-seq append and RecordStore would reject it — that rejection is the correct answer. + const b = new ResidentSession(0, () => undefined); + const sup2 = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s2", + registry: registryFor(b), + }); + const host = await sup2.start(); + b.emit(); + await sup2.restart().catch(() => undefined); + assert.ok(host, "a session was produced"); + await sup.stop(); + await store.close(); + }); +}); + +describe("AgentSupervisor — no process residency", () => { + test("after stop() resolves, the harness process is actually gone", async (t) => { + // The real check. A mock that merely recorded "dispose was called" would pass while the + // process lived on, which is the bug this property exists to prevent. + const child = spawn( + process.execPath, + ["-e", "setInterval(() => {}, 1000)"], + { + stdio: "ignore", + }, + ); + const pid = child.pid; + assert.ok(pid, "spawned a real process to orphan"); + t.after(() => { + try { + process.kill(pid, "SIGKILL"); + } catch { + /* already gone */ + } + }); + + const store = new RecordStore({ dir }); + const session = new ResidentSession(pid, (p) => process.kill(p, "SIGKILL")); + const sup = new AgentSupervisor({ + runtimeId: "fake", + cwd: dir, + store, + sessionId: "s1", + registry: registryFor(session), + }); + await sup.start(); + assert.equal( + alive(pid), + true, + "the process is running while the session is up", + ); + + await sup.stop(); + + assert.equal( + await waitUntilGone(pid), + true, + `no process residency: pid ${pid} must not survive stop()`, + ); + await store.close(); + }); +}); + +/** True while `pid` is still a live process. */ +function alive(pid: number): boolean { + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } +} + +async function waitUntilGone(pid: number, ms = 5000): Promise { + const deadline = Date.now() + ms; + while (Date.now() < deadline) { + if (!alive(pid)) return true; + await new Promise((r) => setTimeout(r, 25)); + } + return !alive(pid); +} diff --git a/docs/ADAPTERS.md b/docs/ADAPTERS.md index 3d2b334..d5d19e5 100644 --- a/docs/ADAPTERS.md +++ b/docs/ADAPTERS.md @@ -49,6 +49,41 @@ for a human at a terminal. Wrong for unattended agents. We are the counter-case, the opposite position — and should document that we disagree with our own dependency, on the record. +## 2. MEASURED, not assumed — re-run with `pnpm probe:runtimes` + +Probed 2026-09-28 against the installed CLIs. This is a **measurement**, not a recollection, and +it is re-runnable so the table cannot silently rot. + +| Runtime | Version | Documents a sandbox | Permission-bypass flag | Self-updates | +| ------- | --------------------- | ------------------- | ---------------------------- | ------------------- | +| claude | 2.1.283 (Claude Code) | no | **yes** | **yes** (`update`) | +| codex | codex-cli 0.157.1 | **yes** (`-s`) | — | **yes** (`update`) | +| grok | grok 1.0.25 | no | **yes** (`--always-approve`) | **yes** (`update`) | +| kimi | 0.26.0 | no | **yes** (`--yolo`) | **yes** (`upgrade`) | +| pi | 0.85.1 | no | — | — | + +**Two things this measurement settles that the prose above only asserted.** + +**1. "Sessions run YOLO by default" is now a verified fact, not a quote.** Four of five CLIs +expose a permission-bypass flag, and oar passes it (`claude --dangerously-skip-permissions`, +`grok --always-approve`, `kimi --yolo`). The premise this whole product is built on is +checkable, and it checks out. Only Codex has a documented sandbox at all. + +**2. FOUR OF FIVE RUNTIMES SELF-UPDATE — and that generalises I15.** + +`k.managed-copy-never-self-upgrades` was recorded for k-carrier: a copy we install and manage +must never also upgrade itself, or there are two writers on one component. That reasoning is not +specific to k-carrier. If Radius ever installs and manages a copy of any of the four CLIs above — +exactly the k-carrier pattern — their built-in `update`/`upgrade` is **the same second writer**. + +So the invariant's scope is wider than one vendored component: **every managed binary on the +host must be upgradeable only by the supervisor.** Recorded in `THREATS.md` as I15, with the +vendor check asserting k-carrier's copy; this probe is how we notice the condition appearing in a +new place. + +**A documented flag is not a working sandbox.** This measures what each CLI _claims_ to support. +The isolation any of them actually enforces is still unmeasured, and `SAFETY.md` §3 says so. + --- ## 3. Capabilities oar exposes diff --git a/docs/MILESTONES.md b/docs/MILESTONES.md index 586735f..60bfea8 100644 --- a/docs/MILESTONES.md +++ b/docs/MILESTONES.md @@ -95,9 +95,36 @@ and all five now fail the suite. A passing suite that survives mutation proves n - [x] oar session binding — `RuntimeRegistry` → `installation` probe → the adapter's `StartSession` (2026-09-28). `src/record-writer.ts` + `src/session-host.ts`, 17 tests. -- [ ] Supervisor lifecycle: start/stop/restart, `agentNoProcessResidency` equivalent -- [ ] k-carrier integration (path dependency vs. built binary — undecided, D-004) -- [ ] `kill -9` mid-turn acceptance test end to end through a real oar session +- [x] Supervisor lifecycle: start/stop/restart, `agentNoProcessResidency` equivalent + (2026-09-28). `src/supervisor.ts`, 7 tests. +- [ ] k-carrier integration (path dependency vs. built binary — D-004; **feasibility now + measured**, see `DECISIONS.md` D-004. A path dependency builds clean; a bare + `cargo build` does not, because 5 of 6 declared targets are not vendored) +- [x] `kill -9` mid-turn acceptance test end to end through a real oar session (2026-09-28) + +**The e2e gate ran against a real Claude session and passed.** `RADIUS_E2E=1` on a machine with +`claude` on PATH: 10 records durable after a real SIGKILL, contiguous from `seq 0`, covering all +three oar record kinds, and replayed identically by a fresh store. The prompt, the `accepted` +response, `system/init`, the model's `assistant` frame and a terminal `result/success` all +survived. The store stamped them with _our_ stream name, not oar's native session id — the +untrusted-name property holding in real conditions, not just in a unit test. + +**And the limit of that result, stated rather than buried.** The run reports which case it hit, +and on this machine the turn _completed_ before the kill — even at a 9s delay. So this proves +**records survive a hard kill**; it does **not** yet prove a kill _during streaming_ is safe. +That case is timing-dependent and not deterministically covered. `RADIUS_E2E_KILL_MS` exists to +aim at it (a cold harness, or a prompt long enough to still be generating), and until someone +captures a run reporting `turn was STILL IN FLIGHT at the kill`, the mid-generation case is +unproven. + +**Supervisor, and the property that matters.** `AgentSupervisor` is deliberately small; the value +is in three properties it makes true, each a way unattended agents go wrong: never two writers on +one stream (`start()` refuses while running — split-brain, one layer up from `never_dual_run`); +nothing observed left unpersisted (`stop()` drains even when `dispose()` throws); and no process +residency. The last is tested for real, not mocked: the test spawns an actual OS process, records +its pid, stops the session, then polls until that pid is gone. A mock asserting "dispose was +called" would pass while the process lived on, which is the bug that property exists to catch. +A wedged harness raises `StopTimeoutError` rather than hanging an unattended host forever. **The session binding's central problem is a type mismatch, not plumbing.** oar delivers records through `RawEventObserver = (record: RawEvent) => void` — synchronous, no await, no backpressure @@ -145,13 +172,29 @@ and CI installs it. Any local development on Node 22 will fail to install the de Follow antiproton's crash-matrix idea: interrupt at every point (before-journal, after-journal, after-action) and assert the invariant holds. -- [ ] Fuzz the kill point across the upgrade state machine -- [ ] Assert `never_dual_run` and `never_bricked` behaviourally, not just by trusting the proofs -- [ ] Test laptop sleep/wake and network loss as first-class cases +- [x] Fuzz the kill point across the write path (2026-09-28) — + `apps/agent-host/test/durability-harness.test.ts` +- [x] Assert the surviving stream is a valid prefix, behaviourally +- [x] Test cursor resume against a truncated stream +- [ ] Assert `never_dual_run` / `never_bricked` for the **upgrade** state machine — blocked on + k-carrier integration (D-004), not on this harness CAUTION: **Do not skip offline.** The cursor contract supports offline-first, but only if we use it that way. A laptop that sleeps mid-run is the normal case, not the edge case. +**What this harness actually asserts, and what it does not.** It truncates the durable file at +_every byte offset_ and asserts what survives is always a contiguous prefix `[0..k]` — never a +gap, never a duplicate, never a half-written record. It also replays a truncated stream and +asserts the resume yields every later record exactly once. + +It does **not** assert "nothing is lost", because that is false and `RecordWriter` documents why: +a record observed but not yet fsync'd dies with the process. What is asserted is the pair that +actually matters to an operator — a corrupt prefix is unacceptable, a lost tail is expected. + +The remaining 0.4 item is the _upgrade_ half, and it is blocked on k-carrier integration rather +than on any missing test. The Lean proofs cover the upgrade transition relation +(`pnpm check:proofs` re-verifies them), but nothing behavioural exercises our use of it yet. + --- ## Phase 1 — Capability broker + sandbox policy @@ -178,10 +221,26 @@ Principal └── createdAt / revokedAt ``` -- [ ] Define in `packages/protocol` -- [ ] **Deliberate omission:** no field holds a raw secret. Ever. If you need one, the design - is wrong. -- [ ] Serialize/deserialize round-trip tests +- [x] Define in `packages/protocol` (2026-09-28) — `src/principal.ts`, 12 tests +- [x] **Deliberate omission:** no field holds a raw secret. Made structural, not conventional +- [x] Serialize/deserialize round-trip tests +- [ ] `decided_by` enforcement wired to a real `approvals` table (phase 2 — needs the DO) + +**The omission is structural, which is the only reason to believe it.** `DATA-MODEL.md` §2 says +in capitals that no column may hold a raw credential, and §4 asks for "a test that scans for it. +Not a code-review convention." So: `Principal` has no `token`/`secret`/`apiKey` field; +`parsePrincipal` **rejects unknown fields**, so a secret cannot ride in through an untyped JSON +bag; and the tests grep serialized principals _and_ grants for nine secret shapes — Anthropic, +OpenAI, GitHub, AWS, Slack, Google, PEM private keys, JWTs, and `api_key=`-style assignments. + +The scanner carries a **positive control** that plants a real-looking AWS key and requires +detection. Without it, a scanner matching nothing would pass every other test in the file and +look convincingly green. + +**Also decided here, and worth writing down:** `CapabilityGrant.expiresAt` is a _required_ field +in the type. `DATA-MODEL.md` §2 marks it "**required.** Default lease 1 hour, per D-008" — so a +permanent grant is now unrepresentable rather than merely discouraged, which is what +`PROTOCOL.md` §6 actually asks for. ### 1.2 — Credential broker @@ -196,6 +255,14 @@ traffic is brokered. The raw credential never enters agent context or memory. - [ ] Loopback only, per-launch nonce, unguessable - [ ] **No secret in any log line, ever** — add a test that greps for it +**NOT STARTED, deliberately.** The proxy, revocation and audit are all buildable today, but real +credential scoping is not: it needs live provider keys for Anthropic/OpenAI/Google/xAI. A broker +that looks finished while credentials escape would be the single worst thing this project could +ship — it makes the pitch true-looking and false, which is the failure mode everything else here +exists to prevent. The contract is now concrete (`CapabilityGrant` in `packages/protocol`), which +is what makes building it safe rather than speculative. It should not be called done until a real +key has been scoped, revoked, and shown unreachable from agent context. + CAUTION: **The one hard invariant, from agent-vault:** a value that must not leak must be _structurally_ unreachable, not merely "we chose not to log it." agent-vault enforces this with an 8-line TTY check — an agent has no TTY, so it _cannot_ run the sensitive command. @@ -227,10 +294,39 @@ native sandbox; per-runtime behavior differs. Budget for it and write the matrix The primitive oar explicitly declines: _"Ownership is the object reference; no in-process lease. Multi-controller arbitration belongs to the application layer."_ -- [ ] One controller per session; leases are the exclusive claim -- [ ] Clock-skew tolerance (leases expire on observation, not on the holder's clock) -- [ ] A new controller can take over a dead one -- [ ] Split-brain is impossible — **test it, don't reason about it** +- [x] One controller per session; leases are the exclusive claim (2026-09-28) +- [x] Clock-skew tolerance (leases expire on observation, not on the holder's clock) +- [x] A new controller can take over a dead one +- [x] Split-brain is impossible — **tested, not reasoned about** +- [x] Wall-clock rollback cannot extend a lease (D-008 Q4) + +`packages/protocol/src/lease.ts`, 13 tests. This is the highest-leverage item in Phase 1, +because it is what turns three _prose_ invariants into enforced code: + +| Invariant | Now enforced by | Was | +| --------- | ------------------------------------------------- | -------------- | +| **I3** | `assertWritable` refuses an expired lease | specified only | +| **I5** | a stale epoch is rejected on every write | specified only | +| **I14** | a rewound `Date.now()` cannot revive a dead lease | specified only | + +`check:invariants` now reports the split honestly — `11 specified only, 3 enforced in code, +1 verified against vendored source` — instead of implying the whole table was prose. + +**Expiry is computed from `process.hrtime`, never `Date.now()`.** The user owns the wall clock; +a lease enforced against it can be extended by one `date` command. The wall clock is recorded in +the anchor for audit and never consulted for expiry. + +**The restart case is the subtle one, and D-008 calls it out by name.** A restart _resets_ the +monotonic clock, so its reading is meaningless across a restart. `LeaseManager.restore` adopts a +persisted anchor — and **drops** any lease whose window already elapsed while the process was +down. Reviving it would let a rewound clock resurrect a dead lease, which is the entire attack +the monotonic design exists to stop. + +**Split-brain is tested, not argued.** The test acquires a lease, lets it expire, takes it over, +then has the _original holder_ — which has no idea it lost anything — attempt five writes. All +five are refused on the epoch. The same test proves a stale holder cannot renew, release, or +extend the new holder's lease. A mutation removing the epoch check, dropping the expiry check, +switching expiry to `Date.now()`, or freezing the epoch counter each fail the suite. ### 1.5 — Invariant checker diff --git a/eslint.config.mjs b/eslint.config.mjs index 5b0c2b6..78cdd2b 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -56,6 +56,7 @@ export default tseslint.config( "./apps/*/tsconfig.json", "./apps/*/tsconfig.test.json", "./packages/*/tsconfig.json", + "./packages/*/tsconfig.test.json", ], tsconfigRootDir: import.meta.dirname, }, diff --git a/package.json b/package.json index bb7cea8..3c69abb 100644 --- a/package.json +++ b/package.json @@ -32,6 +32,7 @@ "check:invariants": "node scripts/check-invariants.mjs", "check:notices": "node scripts/check-notices.mjs", "check:proofs": "node scripts/check-proofs.mjs", + "probe:runtimes": "node scripts/probe-runtimes.mjs", "clean": "turbo run clean && rm -rf node_modules", "prepare": "husky" }, diff --git a/packages/broker/package.json b/packages/broker/package.json new file mode 100644 index 0000000..6281581 --- /dev/null +++ b/packages/broker/package.json @@ -0,0 +1,23 @@ +{ + "name": "@graycode/broker", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "Radius capability broker — scoped tokens, policy, and credential resolution", + "license": "MIT", + "engines": { + "node": ">=24.0.0" + }, + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json && tsc --noEmit -p tsconfig.test.json", + "test": "node --import tsx --test test/*.test.ts" + }, + "dependencies": { + "@graycode/protocol": "workspace:*" + }, + "devDependencies": { + "@types/node": "^24", + "tsx": "^4.21.0", + "typescript": "^5.9.3" + } +} diff --git a/packages/broker/src/policy.ts b/packages/broker/src/policy.ts new file mode 100644 index 0000000..8c01ab0 --- /dev/null +++ b/packages/broker/src/policy.ts @@ -0,0 +1,167 @@ +/** + * Capability decisions, and the audit trail they produce. + * + * Two deliberate properties, both of which are the difference between a control and a comment: + * + * 1. **Deny by default.** `CapabilityPolicy.decide` returns `denied` for anything not explicitly + * granted. There is no "unknown capability → allow" path, and no wildcard that means "all". + * A capability this module has never heard of is refused, which is the only safe default for + * something whose whole purpose is to bound what a compromised agent can reach. + * + * 2. **Every decision is audited, including denials.** Not just grants. A refused capability is + * the most interesting thing that can happen — it is what a prompt injection looks like from + * the inside — so the audit is written on the deny path too. An audit log that only records + * successes is a log that shows you nothing when something goes wrong. + * + * The log records the *decision*, never the credential. `AuditEntry` has no field that could hold + * one, and the secret-shape scan in `test/` asserts it. + */ + +import { isSubset } from "./token.ts"; + +export type Decision = "allowed" | "denied"; + +export interface CapabilityRequest { + readonly principalId: string; + readonly capability: string; + /** What the agent is actually asking for, e.g. `{ repos: ["api"] }`. */ + readonly scope: Readonly>; +} + +export interface Grant { + readonly principalId: string; + readonly capability: string; + readonly scope: Readonly>; + /** A human, always (D-014). An agent approving its own escalation is the thing we forbid. */ + readonly grantedBy: string; + readonly grantedByKind: "human" | "agent" | "service"; + readonly expiresAt: number; +} + +export interface AuditEntry { + readonly at: number; + readonly principalId: string; + readonly capability: string; + readonly requestedScope: Readonly>; + readonly decision: Decision; + /** Why it was refused, when it was. Null on allow. */ + readonly reason: string | null; + readonly launchId: string | null; +} + +export interface DecisionResult { + readonly decision: Decision; + readonly reason: string | null; + readonly audit: AuditEntry; +} + +export class ApprovalNotHumanError extends Error { + constructor( + readonly grantedBy: string, + readonly kind: string, + ) { + super( + `refusing a grant decided by a ${kind} (${grantedBy}). D-014: only a human may approve an ` + + `escalation — an agent able to approve its own is privilege escalation within a tenant.`, + ); + this.name = "ApprovalNotHumanError"; + } +} + +/** + * Deny-by-default capability policy, with an audit trail. + * + * In-memory and synchronous by design, like the lease manager: the logic is the product, and it + * should be provable without a database. The control plane persists `grants`; this decides. + */ +export class CapabilityPolicy { + readonly #grants: Grant[] = []; + readonly #audit: AuditEntry[] = []; + + /** + * Record a grant. Rejects a non-human decider outright (D-014) — this is a check constraint in + * the schema, and enforcing it here too means a caller cannot bypass it by skipping the table. + */ + grant(grant: Grant): void { + if (grant.grantedByKind !== "human") { + throw new ApprovalNotHumanError(grant.grantedBy, grant.grantedByKind); + } + this.#grants.push(grant); + } + + revokeAll(principalId: string): number { + let removed = 0; + for (let i = this.#grants.length - 1; i >= 0; i--) { + const g = this.#grants[i]; + if (g?.principalId === principalId) { + this.#grants.splice(i, 1); + removed += 1; + } + } + return removed; + } + + /** + * Decide one request, and audit the outcome either way. + * + * Order matters: expiry is checked before scope, so an expired grant reports as expired rather + * than as a scope problem. An operator reading the audit trail should see the real reason. + */ + decide( + request: CapabilityRequest, + now: number, + launchId: string | null = null, + ): DecisionResult { + const record = ( + decision: Decision, + reason: string | null, + ): DecisionResult => { + const audit: AuditEntry = { + at: now, + principalId: request.principalId, + capability: request.capability, + requestedScope: { ...request.scope }, + decision, + reason, + launchId, + }; + this.#audit.push(audit); + return { decision, reason, audit }; + }; + + const candidates = this.#grants.filter( + (g) => + g.principalId === request.principalId && + g.capability === request.capability, + ); + if (candidates.length === 0) { + // Deny by default, and say so plainly — "no such capability" is the common case and an + // operator needs to see that it is a refusal, not a lookup miss. + return record( + "denied", + "no grant exists for this principal and capability", + ); + } + + const live = candidates.filter((g) => g.expiresAt > now); + if (live.length === 0) { + return record("denied", "every grant for this capability has expired"); + } + + const permitted = live.some((g) => isSubset(request.scope, g.scope)); + if (!permitted) { + return record("denied", "requested scope exceeds every live grant"); + } + return record("allowed", null); + } + + /** The audit trail, oldest first. */ + entries(): readonly AuditEntry[] { + return this.#audit; + } + + /** Requests refused since start. The number an operator watches. */ + get denialCount(): number { + return this.#audit.filter((e) => e.decision === "denied").length; + } +} diff --git a/packages/broker/src/provider.ts b/packages/broker/src/provider.ts new file mode 100644 index 0000000..0763deb --- /dev/null +++ b/packages/broker/src/provider.ts @@ -0,0 +1,103 @@ +/** + * The provider boundary — and an honest statement of what is and is not verified. + * + * ── THE UNVERIFIED SEAM ───────────────────────────────────────────────────────────────────── + * Everything else in this package is tested. This file is not, and it cannot be, without live + * credentials for Anthropic, OpenAI, Google and xAI plus a real request through each provider's + * auth path. So it is isolated behind a narrow interface, and the isolation is the point: the + * parts that are verified (scoping, revocation, expiry, audit, deny-by-default) do not depend on + * it, and the part that is unverified is small enough to read in one sitting. + * + * **`ResolvedCredential` is deliberately opaque.** A `string` holding a raw API key is one + * careless `JSON.stringify` away from a log line, which is exactly the failure `DATA-MODEL.md` §4 + * and `PROTOCOL.md` §6 forbid. The type has no accessor. The only way to use a credential is to + * hand it back to the `ProviderAdapter` that issued it, so it never becomes a value the rest of + * the program can accidentally serialize. + * + * What a real implementation must still prove, and has not: + * 1. A real provider key is scoped to one principal/launch and cannot act beyond that scope. + * 2. Revoking our token stops the underlying provider call, not just our bookkeeping. + * 3. The credential is not reachable from agent context, a heap dump, a core file, or an + * error message. + * Points 1 and 2 depend on provider features that do not uniformly exist. Until each is + * demonstrated with a live key, this project should not claim it works — see `MILESTONES.md` 1.2. + */ + +/** A live credential. Opaque on purpose: it has no accessor and no `toString`. */ +export interface ResolvedCredential { + readonly __brand: "ResolvedCredential"; + /** Never serialized. Held only for the duration of one upstream call. */ + readonly _opaque: never; +} + +export interface UpstreamRequest { + readonly provider: string; + readonly model: string; + readonly path: string; + readonly body: string; + /** The scoped token this request is authorised under, for the provider to record. */ + readonly tokenHash: string; +} + +export interface UpstreamResponse { + readonly status: number; + readonly body: string; + readonly requestId: string | null; +} + +/** + * Resolves a real credential and performs one upstream call. + * + * The shape is deliberately minimal. A broader interface — one that took arbitrary headers, or + * exposed a general fetch — would let a caller route a credential somewhere the adapter had not + * vetted, which is how brokered secrets escape in practice. + */ +export interface ProviderAdapter { + readonly name: string; + /** + * Whether this provider can actually scope a credential. `false` means a raw key would have to + * be handed over, so the policy layer must refuse rather than pretend the capability exists. + * A provider that reports `true` without honouring it is the worst case in the system, so this + * is a claim to be tested per provider, not trusted. + */ + readonly supportsScopedCredentials: boolean; + /** + * Perform one upstream call using a credential this adapter resolved. The credential is + * returned to the adapter, not handed back to the caller. + */ + call( + request: UpstreamRequest, + credential: ResolvedCredential, + ): Promise; +} + +/** Thrown when a provider cannot honour the scope it was asked for. */ +export class UnscopableProviderError extends Error { + constructor( + readonly provider: string, + readonly capability: string, + ) { + super( + `provider "${provider}" cannot scope a credential for "${capability}". Refusing rather ` + + `than handing over a raw key: a broker that cannot scope is not a broker, and the whole ` + + `guarantee would be a claim rather than a control.`, + ); + this.name = "UnscopableProviderError"; + } +} + +/** + * Gate a capability request on what the provider can actually honour. + * + * This is the check that stops the product from quietly regressing to YOLO: if the provider + * cannot scope, the answer is no. It is deliberately conservative and deliberately strict — a + * capability the adapter has not declared support for is refused. + */ +export function assertScopable( + adapter: ProviderAdapter, + capability: string, +): void { + if (!adapter.supportsScopedCredentials) { + throw new UnscopableProviderError(adapter.name, capability); + } +} diff --git a/packages/broker/src/sandbox.ts b/packages/broker/src/sandbox.ts new file mode 100644 index 0000000..5196156 --- /dev/null +++ b/packages/broker/src/sandbox.ts @@ -0,0 +1,204 @@ +/** + * Per-runtime sandbox policy — deny by default, and honest about what is actually measured. + * + * ── The two things this has to get right ──────────────────────────────────────────────────── + * + * **1. The default is denial.** oar ships YOLO: interactive permission gates disabled, sandboxes + * off, codex at `danger-full-access`. That is right for a human watching a terminal and wrong for + * a scheduled credentialed agent nobody is watching — the only situation Radius targets. So + * every allowance here is an explicit grant, and its absence is a refusal, never a default-allow. + * + * **2. We do not claim isolation we have not measured.** `ADAPTERS.md` says "verify, don't + * assume — this table will rot", and I8 is literally "sandbox strength is measured, or not + * claimed in docs". So `NativeSandbox` has an `unknown` member, `Isolation` has an `unmeasured` + * member, and `claimsIsolation()` returns false for both. That is not hedging — it is what stops + * this module reporting a sandbox for a runtime nobody actually probed, which is precisely how a + * "we sandboxed it" claim turns into fiction. + * + * `unknown` is a legitimate, load-bearing value. A profile that has not been probed must say so, + * and the policy must then refuse rather than assume the friendly answer. + */ + +import type { CapabilityRequest } from "./policy.ts"; + +/** What the runtime's own sandbox provides. `unknown` means "not detected", not "probably fine". */ +export type NativeSandbox = + | "none" + | "danger-full-access" + | "workspace-write" + | "unknown"; + +/** What Radius supplies around it. `unmeasured` means our own boundary is unverified. */ +export type Isolation = "none" | "process-isolation" | "unmeasured"; + +export interface RuntimeSandboxProfile { + readonly runtimeId: string; + /** Detected by probing, never hardcoded. See `ADAPTERS.md` §"verify, don't assume". */ + readonly nativeSandbox: NativeSandbox; + readonly isolation: Isolation; + /** Tools actually observed to be exposed. Empty means none were seen. */ + readonly detectedTools: readonly string[]; + /** When this was probed, so a stale profile is visible rather than silently trusted. */ + readonly probedAt: number; +} + +export type SandboxDecision = + | { readonly kind: "allow"; readonly basis: string } + | { readonly kind: "deny"; readonly reason: string } + | { + readonly kind: "escalate"; + readonly reason: string; + /** Exactly what a human would be asked to approve. */ + readonly request: CapabilityRequest; + }; + +/** An explicit allowance. There is no implicit one. */ +export interface SandboxGrant { + readonly principalId: string; + /** Operation, e.g. `fs:write`, `net:egress`, `exec`, `tool:bash`. */ + readonly operation: string; + /** Optional narrowing, e.g. which paths. Empty scope means the whole operation. */ + readonly scope: Readonly>; + readonly grantedBy: string; + readonly expiresAt: number; +} + +/** + * Whether this profile may claim it is sandboxed — I8, enforced rather than asserted. + * + * Returns false for an unprobed runtime and for a profile whose own isolation is unverified, so + * a caller cannot read "sandboxed: true" out of a module that has not earned it. + */ +export function claimsIsolation(profile: RuntimeSandboxProfile): boolean { + if ( + profile.nativeSandbox === "unknown" || + profile.isolation === "unmeasured" + ) { + return false; + } + return profile.nativeSandbox !== "none" || profile.isolation !== "none"; +} + +/** One line describing the real posture, with no adjective we have not earned. */ +export function describeIsolation(profile: RuntimeSandboxProfile): string { + if (profile.nativeSandbox === "unknown") { + return `${profile.runtimeId}: NOT PROBED — isolation is unverified, so none is claimed`; + } + const native = + profile.nativeSandbox === "none" + ? "no native sandbox" + : `native ${profile.nativeSandbox}`; + const ours = + profile.isolation === "unmeasured" + ? "Radius isolation UNMEASURED" + : profile.isolation === "none" + ? "no Radius isolation" + : `Radius ${profile.isolation}`; + return `${profile.runtimeId}: ${native}, ${ours}`; +} + +export interface SandboxPolicyOptions { + /** Categories that may be escalated to a human. Everything else is simply denied. */ + readonly escalatable?: readonly string[]; +} + +export class SandboxPolicy { + readonly #profiles = new Map(); + readonly #grants: SandboxGrant[] = []; + readonly #escalatable: readonly string[]; + #escalations = 0; + + constructor(options: SandboxPolicyOptions = {}) { + this.#escalatable = options.escalatable ?? [ + "fs:write", + "net:egress", + "exec", + "tool:bash", + ]; + } + + /** Record a probed profile. A second probe replaces the first, so profiles do not go stale. */ + register(profile: RuntimeSandboxProfile): void { + this.#profiles.set(profile.runtimeId, profile); + } + + profile(runtimeId: string): RuntimeSandboxProfile | null { + return this.#profiles.get(runtimeId) ?? null; + } + + grant(grant: SandboxGrant): void { + this.#grants.push(grant); + } + + /** + * Decide one operation. + * + * Order is deliberate: + * 1. An unprobed runtime is refused before anything else. We do not hand an agent a session + * in a sandbox we have never looked at. + * 2. An explicit, unexpired grant allows. + * 3. An escalatable operation escalates — and the caller MUST write an `approvals` row. The + * decision carries the exact request so it cannot be lost in translation. + * 4. Everything else is denied. + */ + decide( + principalId: string, + runtimeId: string, + operation: string, + now: number, + scope: Readonly> = {}, + ): SandboxDecision { + const profile = this.#profiles.get(runtimeId); + if (!profile) { + return { + kind: "deny", + reason: + `runtime "${runtimeId}" has not been probed, so its isolation is unknown. Refusing ` + + `rather than assuming a sandbox we have not seen.`, + }; + } + if (profile.nativeSandbox === "unknown") { + return { + kind: "deny", + reason: + `${describeIsolation(profile)}. Refusing rather than assuming isolation that was ` + + `never measured.`, + }; + } + + const granted = this.#grants.some( + (g) => + g.principalId === principalId && + g.operation === operation && + g.expiresAt > now, + ); + if (granted) { + return { kind: "allow", basis: `explicit grant for ${operation}` }; + } + + const request: CapabilityRequest = { + principalId, + capability: operation, + scope, + }; + if (this.#escalatable.includes(operation)) { + this.#escalations += 1; + return { + kind: "escalate", + reason: + `${operation} is not granted and ${describeIsolation(profile)} cannot be relied on to ` + + `contain it. A human must approve; this must be recorded as an approvals row.`, + request, + }; + } + return { + kind: "deny", + reason: `${operation} is not granted and is not an escalatable category.`, + }; + } + + /** Escalations raised since start — the number an operator watches. */ + escalationCount(): number { + return this.#escalations; + } +} diff --git a/packages/broker/src/token.ts b/packages/broker/src/token.ts new file mode 100644 index 0000000..9cbb57e --- /dev/null +++ b/packages/broker/src/token.ts @@ -0,0 +1,281 @@ +/** + * Scoped token issuance, verification, and revocation. + * + * ── The rule this module exists to enforce ───────────────────────────────────────────────── + * `DATA-MODEL.md` §4: "`broker_token_hash` — store a **hash**. A leaked table must not yield + * working tokens." That is a property of the storage, not a promise about it, so it is built as + * one: **the raw token is returned to the caller once and never retained.** `TokenStore` holds + * only `sha256(token)`. Dumping its internals — or reading its persisted form — yields hashes + * that cannot be replayed, which is exactly the threat (a stolen table) the rule names. + * + * The converse is worth stating too, because it is the part people get wrong: hashing is only + * safe here because the token is 256 bits of CSPRNG output. A hash of a human-chosen secret is a + * cracking target; a hash of random bytes is not. + * + * ── Scope ─────────────────────────────────────────────────────────────────────────────────── + * A token is scoped to a `(principal, launchId, capability)` triple (`DATA-MODEL.md` §4). It is + * time-boxed, revocable without restarting the agent, and narrowable without a restart — so a + * grant can be tightened on a *running* agent, which is the whole reason a broker exists rather + * than a read-only credential handed over at launch. + * + * What is NOT claimed here: that a real provider honours these scopes. That is the provider + * boundary (`provider.ts`) and it stays unverified until a live key is scoped, revoked, and shown + * unreachable from agent context. See `docs/MILESTONES.md` 1.2. + */ + +import { createHash, randomBytes, timingSafeEqual } from "node:crypto"; + +/** What a minted token is allowed to do, and for how long. */ +export interface TokenClaims { + readonly principalId: string; + readonly launchId: string; + readonly capability: string; + readonly scope: Readonly>; + readonly issuedAt: number; + readonly expiresAt: number; + /** Monotonic bound, per D-008 Q4. A rewound wall clock cannot extend this. */ + readonly monotonicBound: number; + /** Per-launch, unguessable. Binds a token to one launch so it cannot be replayed elsewhere. */ + readonly nonce: string; +} + +/** The token plus the hash to persist. The token itself is never stored. */ +export interface MintedToken { + readonly token: string; + /** What `DATA-MODEL.md` calls `broker_token_hash`. Safe to persist. Safe to leak. */ + readonly hash: string; +} + +export type TokenRejection = + | "unknown" + | "revoked" + | "expired" + | "scope-mismatch" + | "launch-mismatch"; + +/** A presented token this store will not honour, and precisely why. */ +export class TokenRejectedError extends Error { + constructor( + readonly reason: TokenRejection, + readonly detail: string, + ) { + super(`token rejected (${reason}): ${detail}`); + this.name = "TokenRejectedError"; + } +} + +interface StoredToken { + claims: TokenClaims; + revokedAt: number | null; +} + +export interface TokenStoreOptions { + /** Monotonic clock for expiry, so a rewound wall clock cannot revive a token. */ + readonly monotonicNow: () => number; + /** CSPRNG. Overridable only so tests can be deterministic. */ + readonly randomBytes?: (n: number) => Buffer; +} + +export interface MintOptions { + readonly principalId: string; + readonly launchId: string; + readonly capability: string; + readonly scope: Readonly>; + /** Milliseconds until expiry, measured on the monotonic clock. */ + readonly ttlMs: number; +} + +/** + * Issues, verifies, narrows, and revokes scoped tokens. + * + * Deliberately in-memory and synchronous. The durable version is a SQLite table keyed on `hash`; + * keeping the logic free of I/O means the security properties can be tested exhaustively here + * rather than trusted in whatever storage the control plane ends up using. + */ +export class TokenStore { + readonly #byHash = new Map(); + readonly #monotonicNow: () => number; + readonly #random: (n: number) => Buffer; + + constructor(options: TokenStoreOptions) { + this.#monotonicNow = options.monotonicNow; + this.#random = options.randomBytes ?? ((n) => randomBytes(n)); + } + + /** + * Issue a scoped token. The raw value is returned here and goes no further — `TokenStore` keeps + * only its hash, so there is no copy left in the heap to steal from a later dump. + */ + mint(options: MintOptions): MintedToken { + for (const [field, value] of Object.entries({ + principalId: options.principalId, + launchId: options.launchId, + capability: options.capability, + })) { + if (typeof value !== "string" || value.length === 0) { + throw new Error(`${field} must be a non-empty string`); + } + } + const now = this.#monotonicNow(); + const token = this.#random(32).toString("base64url"); + const claims: TokenClaims = { + principalId: options.principalId, + launchId: options.launchId, + capability: options.capability, + scope: { ...options.scope }, + issuedAt: now, + expiresAt: now + options.ttlMs, + monotonicBound: now + options.ttlMs, + nonce: this.#random(16).toString("base64url"), + }; + const hash = hashToken(token); + this.#byHash.set(hash, { claims, revokedAt: null }); + return { token, hash }; + } + + /** + * Verify a presented token, returning its claims. + * + * Every failure gets a distinct, named reason because the audit log must distinguish "unknown" + * from "revoked" from "expired" — an operator debugging an outage needs the difference, and + * collapsing them to "denied" would hide a revoked-but-still-circulating token. + * + * Lookup is constant-time. A token is attacker-supplied, and a timing oracle on a lookup key is + * a real leak, however slow. + */ + verify( + token: string, + expected: { launchId?: string; capability?: string } = {}, + ): TokenClaims { + const entry = this.#lookup(token); + if (!entry) { + throw new TokenRejectedError("unknown", "no such token"); + } + if (entry.revokedAt !== null) { + throw new TokenRejectedError( + "revoked", + `revoked at monotonic ${entry.revokedAt}`, + ); + } + if (this.#monotonicNow() >= entry.claims.monotonicBound) { + throw new TokenRejectedError("expired", "past its monotonic bound"); + } + if ( + expected.launchId !== undefined && + expected.launchId !== entry.claims.launchId + ) { + throw new TokenRejectedError( + "launch-mismatch", + "issued for a different launch", + ); + } + if ( + expected.capability !== undefined && + expected.capability !== entry.claims.capability + ) { + throw new TokenRejectedError( + "scope-mismatch", + "issued for a different capability", + ); + } + return entry.claims; + } + + /** + * Narrow a live token's scope in place, without reissuing. + * + * This is what "narrow capabilities without restarting the agent" means operationally: the + * running agent's very next request sees the smaller scope. It can only ever *shrink* — a + * request to widen is refused, because widening is a new decision that belongs in an + * `approvals` row with a human behind it, not in a mutation of a live grant. + */ + narrow(token: string, scope: Readonly>): void { + const entry = this.#lookup(token); + if (!entry) throw new TokenRejectedError("unknown", "no such token"); + if (entry.revokedAt !== null) { + throw new TokenRejectedError("revoked", "cannot narrow a revoked token"); + } + if (!isSubset(scope, entry.claims.scope)) { + throw new Error( + `refusing to widen the scope of a live token. Narrowing is a reduction; widening is a ` + + `new grant and belongs in an approvals row with a human decision behind it.`, + ); + } + entry.claims = { ...entry.claims, scope: { ...scope } }; + } + + /** Revoke. Takes effect on the next request — no agent restart. */ + revoke(token: string): void { + const entry = this.#lookup(token); + if (!entry) throw new TokenRejectedError("unknown", "no such token"); + entry.revokedAt = this.#monotonicNow(); + } + + /** Revoke every token for a launch — the kill switch when a host is lost. */ + revokeLaunch(launchId: string): number { + let count = 0; + const now = this.#monotonicNow(); + for (const entry of this.#byHash.values()) { + if (entry.claims.launchId === launchId && entry.revokedAt === null) { + entry.revokedAt = now; + count += 1; + } + } + return count; + } + + isRevoked(token: string): boolean { + return this.#lookup(token)?.revokedAt != null; + } + + /** Tokens held. For tests and metrics — never exposes a token or a claim. */ + get size(): number { + return this.#byHash.size; + } + + /** Constant-time lookup by token, hashing it first. */ + #lookup(token: string): StoredToken | undefined { + const hash = hashToken(token); + const direct = this.#byHash.get(hash); + if (direct) return direct; + const candidate = Buffer.from(hash, "hex"); + let found: StoredToken | undefined; + for (const [known, entry] of this.#byHash) { + if ( + known.length === hash.length && + timingSafeEqual(Buffer.from(known, "hex"), candidate) + ) { + found = entry; + } + } + return found; + } +} + +/** Hash a token for storage, so persistence layers share one definition. */ +export function hashToken(token: string): string { + return createHash("sha256").update(token, "utf8").digest("hex"); +} + +/** + * True when `narrower` grants no more than `wider`. + * + * Conservative by design: a key in `narrower` must also be in `wider` with a compatible value, + * and anything not understood is treated as *not* a subset — so an unfamiliar scope shape fails + * closed rather than open. + */ +export function isSubset( + narrower: Readonly>, + wider: Readonly>, +): boolean { + for (const [key, value] of Object.entries(narrower)) { + if (!(key in wider)) return false; + const allowed = wider[key]; + if (Array.isArray(allowed) && Array.isArray(value)) { + if (!value.every((v) => allowed.includes(v))) return false; + continue; + } + if (allowed !== value) return false; + } + return true; +} diff --git a/packages/broker/test/broker.test.ts b/packages/broker/test/broker.test.ts new file mode 100644 index 0000000..4d110d5 --- /dev/null +++ b/packages/broker/test/broker.test.ts @@ -0,0 +1,439 @@ +/** + * Broker tests — token lifecycle, deny-by-default policy, and the audit trail. + * + * These cover the parts of 1.2 verifiable without live provider credentials. The provider seam + * (`provider.ts`) is deliberately not exercised; "unscopable providers are refused" is the only + * honest claim to make about it until a real key is in play. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + TokenStore, + TokenRejectedError, + hashToken, + isSubset, +} from "../src/token.ts"; +import { CapabilityPolicy, ApprovalNotHumanError } from "../src/policy.ts"; +import { + assertScopable, + UnscopableProviderError, + type ProviderAdapter, +} from "../src/provider.ts"; + +function store() { + let t = 0; + let seed = 0; + const tokens = new TokenStore({ + monotonicNow: () => t, + // Deterministic bytes so a hash is reproducible here. Production uses the CSPRNG; "the raw + // token is never retained" is asserted structurally, not by inspecting these bytes. + randomBytes: (n) => { + seed += 1; + return Buffer.alloc(n, seed); + }, + }); + return { tokens, advance: (ms: number) => (t += ms) }; +} + +const mintOpts = { + principalId: "pr_1", + launchId: "launch_1", + capability: "repo:read", + scope: { repos: ["api", "web"] }, + ttlMs: 1000, +}; + +const grant = { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api", "web"] }, + grantedBy: "pr_human", + grantedByKind: "human" as const, + expiresAt: 1000, +}; + +function policyWith() { + const policy = new CapabilityPolicy(); + policy.grant(grant); + return policy; +} + +const SECRET_PATTERNS: readonly RegExp[] = [ + /sk-ant-[A-Za-z0-9_-]{8,}/, + /sk-[A-Za-z0-9]{20,}/, + /gh[pousr]_[A-Za-z0-9]{16,}/, + /AKIA[0-9A-Z]{16}/, + /xox[baprs]-[A-Za-z0-9-]{10,}/, + /AIza[0-9A-Za-z_-]{20,}/, + /-----BEGIN [A-Z ]*PRIVATE KEY-----/, + /eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\./, + /"?(?:api[_-]?key|secret|password|passwd|bearer|authorization)"?\s*[:=]\s*"[^"]{8,}"/i, +]; + +describe("TokenStore — storage holds no usable secret", () => { + test("the raw token is never retained; only its hash is", () => { + // DATA-MODEL.md §4: "store a hash. A leaked table must not yield working tokens." The store + // offers no way to read a token back, and the only thing it holds is the hash. + const { tokens } = store(); + const { token, hash } = tokens.mint(mintOpts); + assert.equal( + hash, + hashToken(token), + "the persisted form is exactly sha256(token)", + ); + assert.ok( + !JSON.stringify(tokens).includes(token), + "the raw token must not appear in a serialized store", + ); + }); + + test("two mints for the same claims produce different tokens", () => { + const { tokens } = store(); + const a = tokens.mint(mintOpts); + const b = tokens.mint(mintOpts); + assert.notEqual( + a.token, + b.token, + "a token is never a function of its claims", + ); + assert.notEqual(a.hash, b.hash); + }); +}); + +describe("TokenStore — scope", () => { + test("a token verifies for its own capability and launch", () => { + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + const claims = tokens.verify(token, { + launchId: "launch_1", + capability: "repo:read", + }); + assert.equal(claims.principalId, "pr_1"); + assert.equal(claims.capability, "repo:read"); + }); + + test("a token is refused for a different launch", () => { + // Binding to one launch is what stops a stolen token being replayed onto another host. + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + assert.throws( + () => tokens.verify(token, { launchId: "launch_2" }), + (e: unknown) => + e instanceof TokenRejectedError && e.reason === "launch-mismatch", + ); + }); + + test("a token is refused for a capability it was not issued for", () => { + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + assert.throws( + () => tokens.verify(token, { capability: "net:egress" }), + (e: unknown) => + e instanceof TokenRejectedError && e.reason === "scope-mismatch", + ); + }); + + test("an unknown token is refused", () => { + const { tokens } = store(); + assert.throws( + () => tokens.verify("not-a-real-token"), + (e: unknown) => e instanceof TokenRejectedError && e.reason === "unknown", + ); + }); +}); + +describe("TokenStore — expiry is monotonic", () => { + test("a token expires when monotonic time passes its bound", () => { + const { tokens, advance } = store(); + const { token } = tokens.mint(mintOpts); + advance(1000); + assert.throws( + () => tokens.verify(token), + (e: unknown) => e instanceof TokenRejectedError && e.reason === "expired", + ); + }); +}); + +describe("TokenStore — revocation without restart", () => { + test("revoke takes effect on the next request", () => { + // "Revocable without agent restart" — DATA-MODEL.md §4. The running agent's next call is + // refused and nothing about the agent needs to know it happened. + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + tokens.verify(token); + tokens.revoke(token); + assert.throws( + () => tokens.verify(token), + (e: unknown) => e instanceof TokenRejectedError && e.reason === "revoked", + ); + }); + + test("revokeLaunch kills every token for that launch", () => { + // The lost-laptop kill switch. + const { tokens } = store(); + const a = tokens.mint(mintOpts); + const b = tokens.mint({ ...mintOpts, capability: "net:egress" }); + const other = tokens.mint({ ...mintOpts, launchId: "launch_2" }); + + assert.equal(tokens.revokeLaunch("launch_1"), 2); + for (const t of [a, b]) { + assert.throws( + () => tokens.verify(t.token), + (e: unknown) => + e instanceof TokenRejectedError && e.reason === "revoked", + ); + } + tokens.verify(other.token, { launchId: "launch_2" }); // untouched + }); +}); + +describe("TokenStore — narrowing", () => { + test("a scope can be narrowed on a running token", () => { + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + tokens.narrow(token, { repos: ["api"] }); + assert.deepEqual(tokens.verify(token).scope, { repos: ["api"] }); + }); + + test("widening a live token is refused and leaves it untouched", () => { + // Widening is a new decision, and a new decision needs a human and an approvals row — not a + // mutation of a live grant. + const { tokens } = store(); + const { token } = tokens.mint(mintOpts); + assert.throws( + () => tokens.narrow(token, { repos: ["api", "web", "secrets"] }), + /widen/, + ); + assert.deepEqual( + tokens.verify(token).scope, + { repos: ["api", "web"] }, + "a refused widening changes nothing", + ); + }); +}); + +describe("isSubset — fails closed", () => { + test("a subset is allowed, a superset is not", () => { + assert.equal(isSubset({ repos: ["api"] }, { repos: ["api", "web"] }), true); + assert.equal( + isSubset({ repos: ["secrets"] }, { repos: ["api", "web"] }), + false, + ); + }); + + test("a key absent from the wider scope is not a subset", () => { + assert.equal(isSubset({ unknownKey: 1 }, { repos: [] }), false); + }); + + test("a shape it does not understand is refused, not allowed", () => { + // Fail closed: an unfamiliar scope must never be waved through. + assert.equal( + isSubset({ repos: { nested: true } }, { repos: ["api"] }), + false, + ); + }); +}); + +describe("CapabilityPolicy — deny by default", () => { + test("a capability that was never granted is denied", () => { + // The primary case. A prompt-injected agent asking for something nobody approved is what + // this product exists to contain, and "unknown → allow" is the YOLO default oar ships with. + const r = policyWith().decide( + { principalId: "pr_1", capability: "net:egress", scope: {} }, + 1, + ); + assert.equal(r.decision, "denied"); + assert.match(r.reason ?? "", /no grant exists/); + }); + + test("a granted scope is allowed", () => { + const r = policyWith().decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1, + ); + assert.equal(r.decision, "allowed"); + assert.equal(r.reason, null); + }); + + test("a scope beyond the grant is denied", () => { + const r = policyWith().decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["secrets"] }, + }, + 1, + ); + assert.equal(r.decision, "denied"); + assert.match(r.reason ?? "", /exceeds/); + }); + + test("an expired grant is denied, and reports expiry rather than scope", () => { + // An operator reading the audit trail needs the real reason, not a generic refusal. + const r = policyWith().decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1001, + ); + assert.equal(r.decision, "denied"); + assert.match(r.reason ?? "", /expired/); + }); + + test("another principal's grant does not apply", () => { + const r = policyWith().decide( + { + principalId: "pr_2", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1, + ); + assert.equal(r.decision, "denied", "a grant is per principal, not global"); + }); + + test("revokeAll takes effect immediately", () => { + const policy = policyWith(); + assert.equal(policy.revokeAll("pr_1"), 1); + const r = policy.decide( + { + principalId: "pr_1", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1, + ); + assert.equal(r.decision, "denied"); + }); +}); + +describe("CapabilityPolicy — D-014, only a human may approve", () => { + test("a grant decided by an agent is refused", () => { + // "An agent able to approve its own escalation is privilege escalation within a single + // tenant." Enforced here as well as in the schema, so a caller cannot skip the table. + const policy = new CapabilityPolicy(); + assert.throws( + () => + policy.grant({ + ...grant, + principalId: "pr_agent", + grantedBy: "pr_agent", + grantedByKind: "agent", + }), + ApprovalNotHumanError, + ); + }); + + test("a service cannot approve either", () => { + const policy = new CapabilityPolicy(); + assert.throws( + () => + policy.grant({ + ...grant, + grantedBy: "svc_ci", + grantedByKind: "service", + }), + ApprovalNotHumanError, + ); + }); +}); + +describe("CapabilityPolicy — the audit trail", () => { + test("denials are recorded, not just grants", () => { + // An audit that only records successes shows you nothing when something goes wrong. A + // refused capability is the most interesting event there is — it is what an injection looks + // like from the inside. + const policy = new CapabilityPolicy(); + policy.decide( + { principalId: "pr_1", capability: "net:egress", scope: {} }, + 5, + "launch_1", + ); + assert.equal(policy.entries().length, 1); + assert.equal(policy.entries()[0]?.decision, "denied"); + assert.equal(policy.denialCount, 1); + assert.equal(policy.entries()[0]?.launchId, "launch_1"); + }); + + test("an entry records principal, capability, scope, decision, time and reason", () => { + // "Audit every request: principal, capability, decision, timestamp" — MILESTONES 1.2. + const r = new CapabilityPolicy().decide( + { + principalId: "pr_9", + capability: "repo:read", + scope: { repos: ["api"] }, + }, + 1234, + "launch_7", + ); + assert.equal(r.audit.at, 1234); + assert.equal(r.audit.principalId, "pr_9"); + assert.equal(r.audit.capability, "repo:read"); + assert.deepEqual(r.audit.requestedScope, { repos: ["api"] }); + assert.ok(r.audit.reason); + }); + + test("a serialized audit trail contains no secret-shaped value", () => { + // The log is the thing most likely to ship to a log aggregator, so it is the thing most + // likely to leak. It has no credential field; this proves it. + const policy = new CapabilityPolicy(); + policy.decide( + { + principalId: "pr_1", + capability: "net:egress", + scope: { host: "example.com" }, + }, + 1, + ); + const wire = JSON.stringify(policy.entries()); + for (const pattern of SECRET_PATTERNS) { + assert.ok( + !pattern.test(wire), + `audit trail must not contain a secret (${pattern})`, + ); + } + // Positive control, or a scanner matching nothing would look green. + assert.ok( + SECRET_PATTERNS.some((p) => p.test('{"note":"AKIAIOSFODNN7EXAMPLE"}')), + "the scanner must detect a planted secret, or it is not a check", + ); + }); +}); + +describe("Provider boundary — refuses what it cannot scope", () => { + const adapter = (name: string, scoped: boolean): ProviderAdapter => ({ + name, + supportsScopedCredentials: scoped, + call: () => Promise.reject(new Error("never called in this test")), + }); + + test("a provider that cannot scope is refused rather than given a raw key", () => { + // The check that stops the product quietly regressing to YOLO. If a provider cannot scope, + // the answer is no — handing over a raw key would make every other guarantee a claim. + assert.throws( + () => assertScopable(adapter("some-cli", false), "repo:read"), + UnscopableProviderError, + ); + }); + + test("a provider that declares scoping support passes the gate", () => { + assert.doesNotThrow(() => + assertScopable(adapter("scoped-provider", true), "repo:read"), + ); + }); + + test("the refusal names the provider and the capability", () => { + assert.throws( + () => assertScopable(adapter("some-cli", false), "net:egress"), + /some-cli[\s\S]*net:egress/, + ); + }); +}); diff --git a/packages/broker/test/sandbox.test.ts b/packages/broker/test/sandbox.test.ts new file mode 100644 index 0000000..cf79a91 --- /dev/null +++ b/packages/broker/test/sandbox.test.ts @@ -0,0 +1,193 @@ +/** + * Sandbox policy — deny by default, and never claim isolation that was not measured. + * + * The second half is the one that matters. It is easy to write a module that *behaves* safely + * while its documentation says "sandboxed", and that gap is exactly how an unverified claim + * becomes a published one. So `unknown` and `unmeasured` are first-class values here, and the + * tests below pin that they produce refusals rather than assumptions. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + SandboxPolicy, + claimsIsolation, + describeIsolation, + type RuntimeSandboxProfile, +} from "../src/sandbox.ts"; + +const profile = ( + over: Partial = {}, +): RuntimeSandboxProfile => ({ + runtimeId: "claude", + nativeSandbox: "none", + isolation: "process-isolation", + detectedTools: [], + probedAt: 1000, + ...over, +}); + +const grant = { + principalId: "pr_1", + operation: "net:egress", + scope: {}, + grantedBy: "pr_human", + expiresAt: 2000, +}; + +describe("SandboxPolicy — deny by default", () => { + test("an unprobed runtime is refused", () => { + // We do not hand an agent a session in a sandbox we have never looked at. + const policy = new SandboxPolicy(); + const d = policy.decide("pr_1", "unknown-runtime", "net:egress", 1); + if (d.kind !== "deny") throw new Error(`expected deny, got ${d.kind}`); + assert.match(d.reason, /has not been probed/); + }); + + test("a probed runtime with no grant escalates rather than allowing", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + assert.equal( + policy.decide("pr_1", "claude", "net:egress", 1).kind, + "escalate", + "the default is never allow", + ); + }); + + test("an escalation carries the exact request a human would approve", () => { + // "Escalation writes an approvals row" — the decision has to carry what to write, or the row + // gets lost in translation and the escalation is invisible. + const policy = new SandboxPolicy(); + policy.register(profile()); + const d = policy.decide("pr_1", "claude", "fs:write", 1, { + paths: ["/tmp"], + }); + assert.equal(d.kind, "escalate"); + // assert.equal above already narrows `d` to the escalate variant, so no guard is needed. + assert.deepEqual(d.request, { + principalId: "pr_1", + capability: "fs:write", + scope: { paths: ["/tmp"] }, + }); + }); + + test("a non-escalatable category is denied outright", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + assert.equal( + policy.decide("pr_1", "claude", "ptrace:other-process", 1).kind, + "deny", + "only listed categories may reach a human", + ); + }); + + test("an explicit grant allows", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.grant(grant); + assert.equal( + policy.decide("pr_1", "claude", "net:egress", 1).kind, + "allow", + ); + }); + + test("an expired grant does not allow", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.grant(grant); + assert.notEqual( + policy.decide("pr_1", "claude", "net:egress", 2001).kind, + "allow", + ); + }); + + test("another principal's grant does not apply", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.grant(grant); + assert.notEqual( + policy.decide("pr_2", "claude", "net:egress", 1).kind, + "allow", + ); + }); + + test("escalations are counted", () => { + const policy = new SandboxPolicy(); + policy.register(profile()); + policy.decide("pr_1", "claude", "net:egress", 1); + policy.decide("pr_1", "claude", "fs:write", 2); + assert.equal(policy.escalationCount(), 2); + }); + + describe("SandboxPolicy — I8: measured, or not claimed", () => { + test("an unprobed runtime may not claim isolation", () => { + assert.equal( + claimsIsolation(profile({ nativeSandbox: "unknown" })), + false, + ); + }); + + test("unmeasured Radius isolation may not claim isolation", () => { + // Our own boundary being unverified is just as disqualifying as the runtime's. + assert.equal( + claimsIsolation( + profile({ nativeSandbox: "none", isolation: "unmeasured" }), + ), + false, + ); + }); + + test("a runtime with no isolation at all does not claim it", () => { + assert.equal( + claimsIsolation(profile({ nativeSandbox: "none", isolation: "none" })), + false, + ); + }); + + test("a probed runtime with a real boundary does claim it", () => { + assert.equal(claimsIsolation(profile()), true); + }); + + test("an unprobed runtime is refused at decision time, not only at claim time", () => { + // The description is for humans; this is the control. An unknown profile must not become an + // allow merely by not being described anywhere. + const policy = new SandboxPolicy(); + policy.register(profile({ nativeSandbox: "unknown" })); + const d = policy.decide("pr_1", "claude", "net:egress", 1); + if (d.kind !== "deny") throw new Error(`expected deny, got ${d.kind}`); + assert.match(d.reason, /never measured/); + }); + + test("describeIsolation never uses an adjective that was not earned", () => { + assert.match( + describeIsolation(profile({ nativeSandbox: "unknown" })), + /NOT PROBED/, + "an unprobed runtime says so plainly", + ); + assert.match( + describeIsolation(profile({ isolation: "unmeasured" })), + /UNMEASURED/, + "unverified isolation of our own is stated, not hidden", + ); + }); + }); + + describe("SandboxPolicy — profiles do not go stale", () => { + test("a re-probe replaces the previous profile", () => { + // ADAPTERS.md: "verify, don't assume — this table will rot". A re-probe must actually take + // effect, or the stale one is trusted forever. + const policy = new SandboxPolicy(); + policy.register(profile({ nativeSandbox: "none", probedAt: 1 })); + policy.register( + profile({ nativeSandbox: "workspace-write", probedAt: 2 }), + ); + assert.equal(policy.profile("claude")?.nativeSandbox, "workspace-write"); + assert.equal(policy.profile("claude")?.probedAt, 2); + }); + + test("an unprobed runtime has no profile", () => { + assert.equal(new SandboxPolicy().profile("nope"), null); + }); + }); +}); diff --git a/packages/broker/tsconfig.json b/packages/broker/tsconfig.json new file mode 100644 index 0000000..4782c91 --- /dev/null +++ b/packages/broker/tsconfig.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "outDir": "dist", + "rootDir": "src", + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts"] +} diff --git a/packages/broker/tsconfig.test.json b/packages/broker/tsconfig.test.json new file mode 100644 index 0000000..b9039d2 --- /dev/null +++ b/packages/broker/tsconfig.test.json @@ -0,0 +1,12 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "noEmit": true, + "declaration": false, + "sourceMap": false, + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts", "test/**/*.ts"] +} diff --git a/packages/protocol/package.json b/packages/protocol/package.json new file mode 100644 index 0000000..b3f0ae1 --- /dev/null +++ b/packages/protocol/package.json @@ -0,0 +1,20 @@ +{ + "name": "@graycode/protocol", + "version": "0.0.1", + "private": true, + "type": "module", + "description": "Radius protocol — principals, capabilities, and lease semantics", + "license": "MIT", + "engines": { + "node": ">=24.0.0" + }, + "scripts": { + "typecheck": "tsc --noEmit -p tsconfig.json && tsc --noEmit -p tsconfig.test.json", + "test": "node --import tsx --test test/*.test.ts" + }, + "devDependencies": { + "@types/node": "^24", + "tsx": "^4.21.0", + "typescript": "^5.9.3" + } +} diff --git a/packages/protocol/src/approvals.ts b/packages/protocol/src/approvals.ts new file mode 100644 index 0000000..25192ee --- /dev/null +++ b/packages/protocol/src/approvals.ts @@ -0,0 +1,147 @@ +/** + * The approvals table — the one place Radius claims **exactly-once** delivery. + * + * D-013 and `PROTOCOL.md` §4 both say the same thing: `approvals.request_id` carries a unique + * index, so a replayed decision is an idempotent no-op *by construction*. Everything else on the + * wire is at-least-once with idempotent consumers, and the docs are emphatic that the difference + * must not be blurred — "do not put 'exactly-once' in the marketing for anything but approvals". + * + * So the mechanism is small and the property is structural: a decision is keyed by a + * client-supplied `requestId`, and replaying one returns the *original* decision rather than + * recording a second one. A genuinely conflicting replay — same `requestId`, different content — + * is refused rather than silently overwritten, because "the human approved X but the replay says + * Y" is exactly the bug a unique index exists to surface. + */ + +export type ApprovalOutcome = "approved" | "denied"; + +export interface ApprovalRequest { + /** Client-supplied UUID. The idempotency key, and the entire mechanism behind exactly-once. */ + readonly requestId: string; + readonly principalId: string; + readonly capability: string; + readonly scope: Readonly>; + readonly requestedAt: number; +} + +export interface ApprovalDecision { + readonly requestId: string; + readonly outcome: ApprovalOutcome; + /** MUST be a human. D-014: an agent approving its own escalation is the attack. */ + readonly decidedBy: string; + readonly decidedByKind: "human" | "agent" | "service"; + readonly decidedAt: number; + /** Free-text reason. Required on a denial, so a refusal is never anonymous. */ + readonly reason: string | null; +} + +export type SubmitResult = + | { readonly kind: "recorded"; readonly decision: ApprovalDecision } + | { + /** Same requestId, same content — a client retry. The original decision is returned. */ + readonly kind: "replayed"; + readonly decision: ApprovalDecision; + }; + +/** The same requestId was replayed with different content. Never silently overwrite. */ +export class ApprovalConflictError extends Error { + constructor(readonly requestId: string) { + super( + `request "${requestId}" was replayed with different content. A unique index would reject ` + + `this, and so does this: overwriting would mean an agent could re-approve a decision it ` + + `was not granted.`, + ); + this.name = "ApprovalConflictError"; + } +} + +/** A non-human tried to decide. D-014. */ +export class NonHumanDecisionError extends Error { + constructor( + readonly requestId: string, + readonly kind: string, + ) { + super( + `approval "${requestId}" was decided by a ${kind}. Only a human may approve an escalation — ` + + `an agent able to approve its own is privilege escalation within a single tenant (D-014).`, + ); + this.name = "NonHumanDecisionError"; + } +} + +/** A denial with no reason. A refusal an operator cannot explain is not a usable refusal. */ +export class MissingDenialReasonError extends Error { + constructor(readonly requestId: string) { + super( + `approval "${requestId}" was denied with no reason. A refusal nobody can explain is not a ` + + `usable refusal — record why.`, + ); + this.name = "MissingDenialReasonError"; + } +} + +function sameRequest(a: ApprovalRequest, b: ApprovalRequest): boolean { + return ( + a.principalId === b.principalId && + a.capability === b.capability && + JSON.stringify(a.scope) === JSON.stringify(b.scope) + ); +} + +export class ApprovalsStore { + readonly #byRequestId = new Map< + string, + { request: ApprovalRequest; decision: ApprovalDecision } + >(); + + /** + * Record a decision, or return the original if this is a client retry. + * + * This is the exactly-once claim, and it is a property of the data structure rather than of + * how carefully the caller behaves. Two concurrent retries of the same `requestId` cannot + * produce two rows, because there is one slot keyed by it. + */ + submit(request: ApprovalRequest, decision: ApprovalDecision): SubmitResult { + if (decision.decidedByKind !== "human") { + throw new NonHumanDecisionError( + request.requestId, + decision.decidedByKind, + ); + } + if (decision.outcome === "denied" && !decision.reason) { + throw new MissingDenialReasonError(request.requestId); + } + + const existing = this.#byRequestId.get(request.requestId); + if (existing) { + if (!sameRequest(existing.request, request)) { + throw new ApprovalConflictError(request.requestId); + } + // Same request, same id: a retry. Return what was already decided, unchanged. + return { kind: "replayed", decision: existing.decision }; + } + + this.#byRequestId.set(request.requestId, { request, decision }); + return { kind: "recorded", decision }; + } + + get(requestId: string): ApprovalDecision | null { + return this.#byRequestId.get(requestId)?.decision ?? null; + } + + has(requestId: string): boolean { + return this.#byRequestId.has(requestId); + } + + /** Total decisions recorded. A replay must not increase this. */ + get size(): number { + return this.#byRequestId.size; + } + + entries(): readonly { + request: ApprovalRequest; + decision: ApprovalDecision; + }[] { + return [...this.#byRequestId.values()]; + } +} diff --git a/packages/protocol/src/budget.ts b/packages/protocol/src/budget.ts new file mode 100644 index 0000000..209a53f --- /dev/null +++ b/packages/protocol/src/budget.ts @@ -0,0 +1,197 @@ +/** + * Hard spend caps — the Phase 4 gate, built as pure logic. + * + * The gate reads: *"A hard spend cap cannot be exceeded, including under retry and partial + * failure. Demonstrated under fault injection."* The second sentence is the whole difficulty, + * and the reason this module exists rather than an `if (spent > cap)` at the call site: + * + * - **Retries cost twice.** A provider that fails after accepting a request has already billed + * it. A cap counting only *successful* turns overshoots under exactly the conditions where an + * operator is retrying hardest. + * - **Partial failure splits reservation from settlement.** A crash between the two must not + * leak a reservation forever, nor release headroom already spent. + * - **Concurrent turns race.** Two in-flight turns must not both see "under cap" and together + * cross it. The cap is enforced at *reservation* time against committed + reserved, never + * recomputed from a running total at settlement. + * + * The invariant, stated once: **`committed + reserved` never exceeds `cap`.** Everything here + * exists to hold that, and the tests assert it directly rather than asserting an outcome that + * merely implies it. + */ + +import { ProtocolError } from "./principal.ts"; + +export interface Budget { + /** Hard ceiling in the smallest currency unit, to avoid float drift. */ + readonly capMinor: number; + readonly currency: string; +} + +export class BudgetExceededError extends Error { + constructor( + readonly requestedMinor: number, + readonly committedMinor: number, + readonly reservedMinor: number, + readonly capMinor: number, + ) { + super( + `spend cap exceeded: committing ${requestedMinor} would take committed+reserved to ` + + `${committedMinor + reservedMinor + requestedMinor}, over the cap of ${capMinor}. ` + + `Refusing the reservation.`, + ); + this.name = "BudgetExceededError"; + } +} + +export interface Reservation { + readonly id: string; + readonly principalId: string; + /** What we set aside: the upper bound, so a partial result cannot overshoot. */ + readonly reservedMinor: number; + readonly createdAt: number; +} + +interface Tracked extends Reservation { + settledMinor: number | null; +} + +export class CostLedger { + readonly #budget: Budget; + readonly #committed = new Map(); + readonly #reservations = new Map(); + + constructor(budget: Budget) { + if (!Number.isInteger(budget.capMinor) || budget.capMinor < 0) { + throw new ProtocolError("capMinor must be a non-negative integer"); + } + this.#budget = budget; + } + + committedFor(principalId: string): number { + return this.#committed.get(principalId) ?? 0; + } + + /** + * Headroom currently held back by *unsettled* reservations. + * + * A settled reservation is deliberately excluded: its money has already moved into + * `committed`, so counting it here as well would charge for it twice and make `available()` + * understate the budget — silently throttling an agent that still has headroom. + */ + reservedFor(principalId: string): number { + let total = 0; + for (const r of this.#reservations.values()) { + if (r.principalId === principalId && r.settledMinor === null) { + total += r.reservedMinor; + } + } + return total; + } + + /** + * Reserve headroom before spending. + * + * The check is against `committed + reserved + requested`, so two concurrent turns cannot both + * see headroom and collectively cross the cap. Reserving the *upper* bound is what makes a + * partial result harmless: settlement only ever moves money from reserved to committed. + */ + reserve( + principalId: string, + upperBoundMinor: number, + now: number, + ): Reservation { + if (!Number.isInteger(upperBoundMinor) || upperBoundMinor < 0) { + throw new ProtocolError("upperBoundMinor must be a non-negative integer"); + } + const committed = this.committedFor(principalId); + const reserved = this.reservedFor(principalId); + if (committed + reserved + upperBoundMinor > this.#budget.capMinor) { + throw new BudgetExceededError( + upperBoundMinor, + committed, + reserved, + this.#budget.capMinor, + ); + } + const reservation: Tracked = { + id: `res_${principalId}_${now}_${this.#reservations.size}`, + principalId, + reservedMinor: upperBoundMinor, + createdAt: now, + settledMinor: null, + }; + this.#reservations.set(reservation.id, reservation); + return reservation; + } + + /** + * Settle a reservation at its actual cost. + * + * Actual may be *less* than reserved (headroom returns to the budget) or equal. It may not be + * more: if the real cost exceeds the bound we reserved, the cap has already been crossed, and + * the honest thing is to say so rather than quietly over-committing. + */ + settle(reservationId: string, actualMinor: number): number { + const reservation = this.#reservations.get(reservationId); + if (!reservation) { + throw new ProtocolError(`no such reservation: ${reservationId}`); + } + if (reservation.settledMinor !== null) { + // Settling twice is a retry, not an error, and must not double-charge. + return reservation.settledMinor; + } + if (actualMinor < 0) { + throw new ProtocolError("actualMinor must be non-negative"); + } + if (actualMinor > reservation.reservedMinor) { + throw new BudgetExceededError( + actualMinor - reservation.reservedMinor, + this.committedFor(reservation.principalId), + this.reservedFor(reservation.principalId), + this.#budget.capMinor, + ); + } + reservation.settledMinor = actualMinor; + this.#committed.set( + reservation.principalId, + this.committedFor(reservation.principalId) + actualMinor, + ); + return actualMinor; + } + + /** + * Release a reservation that never became a call. + * + * Distinct from `settle(0)`: a release means no money moved at all, so it must not create a + * committed row. Without this, a crash between reserve and send would leak headroom forever and + * the agent would stop working with budget still available. + */ + release(reservationId: string): void { + const reservation = this.#reservations.get(reservationId); + if (!reservation) + throw new ProtocolError(`no such reservation: ${reservationId}`); + if (reservation.settledMinor === null) + this.#reservations.delete(reservationId); + } + + /** Headroom still spendable, accounting for in-flight reservations. */ + available(principalId: string): number { + return Math.max( + 0, + this.#budget.capMinor - + this.committedFor(principalId) - + this.reservedFor(principalId), + ); + } + + get capMinor(): number { + return this.#budget.capMinor; + } + + get openReservations(): number { + let n = 0; + for (const r of this.#reservations.values()) + if (r.settledMinor === null) n += 1; + return n; + } +} diff --git a/packages/protocol/src/index.ts b/packages/protocol/src/index.ts new file mode 100644 index 0000000..a50d7c5 --- /dev/null +++ b/packages/protocol/src/index.ts @@ -0,0 +1,84 @@ +/** + * `@graycode/protocol` — the stable surface. See `PROTOCOL.md` for the wire contract. + * + * Phase 1.1 (Principal) and 1.4 (leases). Deliberately small: these are the two pieces whose + * logic can be proved exhaustively without a platform runtime, and getting them right is what + * makes the broker's contract concrete instead of speculative. + * + * `export type` is used explicitly for every type, because `verbatimModuleSyntax` is on and a + * value-style re-export of a type is a compile error rather than a silent runtime no-op. + */ + +export type { + CapabilityGrant, + CreatePrincipalOptions, + Principal, + PrincipalKind, +} from "./principal.ts"; +export { + ProtocolError, + createPrincipal, + isRevoked, + parsePrincipal, + serializePrincipal, +} from "./principal.ts"; + +export type { + Lease, + LeaseManagerOptions, + MonotonicAnchor, + MonotonicClockOptions, + MonotonicSource, +} from "./lease.ts"; +export { + DEFAULT_LEASE_TTL_MS, + LeaseExpiredError, + LeaseHeldError, + LeaseManager, + MonotonicClock, + StaleEpochError, + hrtimeSource, +} from "./lease.ts"; + +export type { + ApprovalDecision, + ApprovalOutcome, + ApprovalRequest, + SubmitResult, +} from "./approvals.ts"; +export { + ApprovalConflictError, + ApprovalsStore, + MissingDenialReasonError, + NonHumanDecisionError, +} from "./approvals.ts"; + +export type { Budget, Reservation } from "./budget.ts"; +export { BudgetExceededError, CostLedger } from "./budget.ts"; + +export type { Cursor, OutboxEntry, Tenant, TenantResolver } from "./sync.ts"; +export { + CursorMerger, + Outbox, + StaticTenantTable, + UnresolvedTenantError, +} from "./sync.ts"; + +export type { + ChangeClass, + Envelope, + ParsedVersion, + ReleaseNote, +} from "./version.ts"; +export { + PROTOCOL_MAJOR, + PROTOCOL_MINOR, + PROTOCOL_VERSION, + BreakingReleaseError, + UnsupportedVersionError, + assertCompatible, + assertVersionBump, + classifyRelease, + formatVersion, + parseVersion, +} from "./version.ts"; diff --git a/packages/protocol/src/lease.ts b/packages/protocol/src/lease.ts new file mode 100644 index 0000000..9f96a7d --- /dev/null +++ b/packages/protocol/src/lease.ts @@ -0,0 +1,280 @@ +/** + * Monotonic time, and the lease manager built on it. + * + * oar declines this outright: *"Ownership is the object reference; no in-process lease. + * Multi-controller arbitration belongs to the application layer."* That sentence is the company — + * the layer nobody else is building. This is it. + * + * ── Why a plain `Date.now()` lease is not good enough ─────────────────────────────────────── + * Because the user controls the clock. A lease enforced against wall time can be extended + * indefinitely by setting the system clock back — not a theoretical attack, one `date` command + * away on the machine the agent runs on. D-008 Q4 settles it: **bind leases against monotonic + * time.** Expiry is computed from `process.hrtime`, which never moves backwards and which the + * user cannot reach. + * + * The subtlety D-008 flags, implemented here: a restart **resets** the monotonic clock, so its + * reading is meaningless across a restart unless re-anchored against persisted state. That is + * what `MonotonicClock.restore` and `LeaseManager.restore` exist for, and why a lease whose + * window elapsed while the process was down is dropped rather than revived. + * + * ── The second property: fencing ─────────────────────────────────────────────────────────── + * Expiry alone does not prevent split-brain. A partitioned holder does not notice its lease + * expired; it keeps writing. So every lease carries an `epoch` that increments on takeover, and + * **every write is checked against it** (D-012, invariant I5). A stale holder is rejected no + * matter what it believes about its own lease. This is the difference between "we have leases" + * and "we cannot split-brain". + */ + +/** A monotonic millisecond reading. Never decreases within a process. */ +export interface MonotonicSource { + now(): number; +} + +/** `process.hrtime`-backed. Injected in tests so time can be controlled. */ +export const hrtimeSource: MonotonicSource = { + now: () => Number(process.hrtime.bigint() / 1_000_000n), +}; + +/** What is persisted so a restart can re-anchor the monotonic clock. */ +export interface MonotonicAnchor { + /** Monotonic reading at the moment of anchoring. */ + readonly monotonicAt: number; + /** Wall clock at the same instant. Recorded for audit, never trusted for expiry. */ + readonly wallAt: number; +} + +export interface MonotonicClockOptions { + readonly source?: MonotonicSource; + readonly wallNow?: () => number; + /** State persisted by a previous process, if any. */ + readonly anchor?: MonotonicAnchor; +} + +/** + * A monotonic clock that survives a process restart. + * + * Deliberately narrow: call `now()` for every expiry decision, `anchor()` to persist. Never + * compare `now()` against a `Date.now()` value. + */ +export class MonotonicClock { + readonly #source: MonotonicSource; + readonly #wallNow: () => number; + /** The persisted monotonic value we are measuring from. */ + #anchorMonotonic: number; + /** This process's source reading at the moment we adopted `#anchorMonotonic`. */ + #sourceAtAnchor: number; + + constructor(options: MonotonicClockOptions = {}) { + this.#source = options.source ?? hrtimeSource; + this.#wallNow = options.wallNow ?? Date.now; + this.#anchorMonotonic = options.anchor?.monotonicAt ?? 0; + this.#sourceAtAnchor = this.#source.now(); + } + + /** + * Restore a persisted anchor, so time is continuous across a restart. + * + * After a restart the process monotonic clock has reset to ~0. Adopting the anchor means + * `now()` *starts* at the persisted value and advances from there — so a lease that was live + * before the bounce is still live, and one whose window elapsed while we were down is not. + * + * Note what this does NOT do: it never consults the wall clock. That is the whole point. A + * user who rewinds `Date.now()` cannot move this reading, and therefore cannot revive a lease. + */ + restore(anchor: MonotonicAnchor): void { + this.#anchorMonotonic = anchor.monotonicAt; + this.#sourceAtAnchor = this.#source.now(); + } + + /** Monotonic milliseconds. Non-decreasing for the lifetime of the clock. */ + now(): number { + return this.#anchorMonotonic + (this.#source.now() - this.#sourceAtAnchor); + } + + /** Persist this. Store it; feed it back via `restore` on the next start. */ + anchor(): MonotonicAnchor { + return { monotonicAt: this.now(), wallAt: this.#wallNow() }; + } +} + +export interface Lease { + readonly resource: string; + readonly holderId: string; + readonly acquiredAt: number; + readonly expiresAt: number; + /** Fencing token. Increments on every takeover; checked on every write. */ + readonly epoch: number; +} + +/** The resource is already leased by a live holder. */ +export class LeaseHeldError extends Error { + constructor( + readonly resource: string, + readonly holderId: string, + ) { + super( + `"${resource}" is held by ${holderId}. One controller per resource is the whole point; ` + + `if you meant to take over, wait for expiry or release it explicitly.`, + ); + this.name = "LeaseHeldError"; + } +} + +/** The presented epoch is not current. A partitioned holder lands here. */ +export class StaleEpochError extends Error { + constructor( + readonly resource: string, + readonly presented: number, + readonly current: number, + ) { + super( + `stale epoch ${presented} for "${resource}" (current is ${current}). This holder was ` + + `superseded — a takeover happened while it was partitioned. Refusing the write.`, + ); + this.name = "StaleEpochError"; + } +} + +/** The lease has expired on the authority's clock. */ +export class LeaseExpiredError extends Error { + constructor( + readonly resource: string, + readonly holderId: string, + ) { + super( + `the lease on "${resource}" held by ${holderId} has expired. Expiry is evaluated on the ` + + `authority's monotonic clock, never the holder's, so clock skew cannot revive it.`, + ); + this.name = "LeaseExpiredError"; + } +} + +/** + * Exclusive, epoch-fenced, monotonic leases. One controller per resource. + * + * Pure and synchronous by design: the real implementation runs inside a Durable Object, where + * single-threaded execution *is* the mutual exclusion. Keeping the logic free of I/O means it + * can be tested exhaustively here rather than trusted inside a platform runtime. + */ +export class LeaseManager { + readonly #clock: MonotonicClock; + readonly #ttlMs: number; + readonly #leases = new Map(); + + constructor(options: LeaseManagerOptions) { + this.#clock = options.clock; + this.#ttlMs = options.ttlMs ?? DEFAULT_LEASE_TTL_MS; + } + + /** + * Claim a resource exclusively, or take over an expired one. + * + * On takeover the epoch **increments**. That is the fencing token: any write the previous holder + * attempts with its old epoch is rejected from that moment on, however confidently it believes + * it still holds the lease. + */ + acquire(resource: string, holderId: string): Lease { + const now = this.#clock.now(); + const existing = this.#leases.get(resource); + + if (existing && existing.expiresAt > now) { + throw new LeaseHeldError(resource, existing.holderId); + } + + const lease: Lease = { + resource, + holderId, + acquiredAt: now, + expiresAt: now + this.#ttlMs, + epoch: (existing?.epoch ?? 0) + 1, + }; + this.#leases.set(resource, lease); + return lease; + } + + /** + * The authority's view of a resource. `null` when free or expired — an expired lease is not a + * lease, and reporting it as one is how a second controller talks itself into thinking it may + * write. + */ + current(resource: string): Lease | null { + const lease = this.#leases.get(resource); + if (!lease) return null; + return lease.expiresAt > this.#clock.now() ? lease : null; + } + + /** + * Assert a holder may write. Call this on EVERY write — that is the entire point of the epoch. + * + * Checks in order: the lease exists, the epoch is current, it has not expired, and the holder + * is the one who took it. A caller that skips this check is a caller that can split-brain. + */ + assertWritable(resource: string, holderId: string, epoch: number): Lease { + const lease = this.#leases.get(resource); + if (!lease) { + throw new LeaseExpiredError(resource, holderId); + } + if (lease.epoch !== epoch) { + throw new StaleEpochError(resource, epoch, lease.epoch); + } + if (lease.expiresAt <= this.#clock.now()) { + throw new LeaseExpiredError(resource, holderId); + } + if (lease.holderId !== holderId) { + throw new LeaseHeldError(resource, lease.holderId); + } + return lease; + } + + /** Extend a live lease. Requires a current epoch — a stale holder cannot extend. */ + renew(resource: string, holderId: string, epoch: number): Lease { + const lease = this.assertWritable(resource, holderId, epoch); + const extended: Lease = { + ...lease, + expiresAt: this.#clock.now() + this.#ttlMs, + }; + this.#leases.set(resource, extended); + return extended; + } + + /** Give the lease up early. A new holder may then take over without waiting for expiry. */ + release(resource: string, holderId: string, epoch: number): void { + this.assertWritable(resource, holderId, epoch); + this.#leases.delete(resource); + } + + /** Snapshot for persistence. */ + entries(): readonly Lease[] { + return [...this.#leases.values()]; + } + + /** + * Rehydrate leases persisted by a previous process. + * + * Exists because a restart resets the monotonic clock, and a lease that was live must still be + * live afterwards. Two rules, and the second is the one that matters: + * + * 1. The monotonic anchor is restored too, or `now()` resets to ~0 and every live lease looks + * centuries old — a restart would expire everything. + * 2. **A lease whose window already elapsed is dropped, not revived.** D-008 says this in as + * many words: "if the persisted state shows the monotonic window has already elapsed, the + * lease is expired regardless of the wall clock." Rehydrating it would let a rewound clock + * resurrect a dead lease, which is the entire attack Q4 exists to stop. + */ + restore(leases: readonly Lease[], anchor: MonotonicAnchor): void { + this.#clock.restore(anchor); + const now = this.#clock.now(); + for (const lease of leases) { + if (lease.expiresAt <= now) continue; // elapsed while we were down + this.#leases.set(lease.resource, lease); + } + } +} + +export interface LeaseManagerOptions { + readonly clock: MonotonicClock; + /** Milliseconds a lease is held. Default 1 hour, per D-008. */ + readonly ttlMs?: number; +} + +export const DEFAULT_LEASE_TTL_MS = 60 * 60 * 1000; diff --git a/packages/protocol/src/principal.ts b/packages/protocol/src/principal.ts new file mode 100644 index 0000000..5a463f8 --- /dev/null +++ b/packages/protocol/src/principal.ts @@ -0,0 +1,156 @@ +/** + * The Principal — "the core type. Everything else derives from it." (`MILESTONES.md` 1.1) + * + * A principal has a center and a radius: a stable identity, the workspace it belongs to, and the + * set of capabilities it may exercise. Everything in the protocol is expressed in these terms. + * + * ── The deliberate omission, made structural ────────────────────────────────────────────── + * `DATA-MODEL.md` §2 says, in capitals: "No column in any table holds a raw credential." §4 adds + * that `broker_token_hash` is a *hash*, and that raw credentials live only inside the broker + * process, in memory, unreferenced by name in any durable structure. + * + * The usual enforcement is a code-review convention, which is a comment. So the omission here + * is structural instead: **this type has no field that can hold a secret, and there is no way to + * smuggle one in.** There is no `token`, no `secret`, no `apiKey` — and `parsePrincipal` rejects + * unknown fields outright, so a secret cannot ride along in an untyped JSON bag. + * + * That is still a convention enforced by a compiler rather than by a type system, so it is not + * taken on trust: `test/secrets.test.ts` serializes real principals and grants and greps the bytes + * for secret-shaped values, which is the mechanism `DATA-MODEL.md` §4 asks for by name. + */ + +/** Who a principal is. D-014: only a human may approve an escalation. */ +export type PrincipalKind = "human" | "agent" | "service"; + +export interface Principal { + /** Stable identity. */ + readonly id: string; + readonly kind: PrincipalKind; + /** The workspace/task this principal belongs to — _the radius_. */ + readonly centerId: string; + readonly displayName: string; + readonly createdAt: number; + /** Soft revocation, for audit (`DATA-MODEL.md` §2). `null` means live. */ + readonly revokedAt: number | null; +} + +/** + * One granted capability. `expiresAt` is **required, never optional**: `DATA-MODEL.md` §2 marks + * it "**required.** Default lease 1 hour, per D-008", and a capability that cannot expire is + * exactly the permanent grant `PROTOCOL.md` §6 forbids. Making it required in the type means a + * permanent grant is unrepresentable rather than merely discouraged. + */ +export interface CapabilityGrant { + readonly capability: string; + /** JSON describing which repos/hosts/paths. Empty scope means "nothing granted". */ + readonly scope: Readonly>; + /** Every grant traces to a human decision (`DATA-MODEL.md` §2, D-014). */ + readonly grantedBy: string; + readonly grantedAt: number; + readonly expiresAt: number; + /** Monotonic window, per D-008 Q4. Wall clock alone can be rolled back. */ + readonly monotonicBound: number; + readonly approvalId: string; +} + +export class ProtocolError extends Error { + constructor(message: string) { + super(message); + this.name = "ProtocolError"; + } +} + +const KINDS: readonly PrincipalKind[] = ["human", "agent", "service"]; + +function assertNonEmptyString(value: unknown, field: string): string { + if (typeof value !== "string" || value.length === 0) { + throw new ProtocolError(`${field} must be a non-empty string`); + } + return value; +} + +function assertFiniteNumber(value: unknown, field: string): number { + if (typeof value !== "number" || !Number.isFinite(value)) { + throw new ProtocolError(`${field} must be a finite number`); + } + return value; +} + +/** + * Parse and validate a principal from untrusted JSON. + * + * Unknown fields are **rejected**, not ignored. That is the second half of the structural + * omission above: a permissive parser would accept `{"id":..., "apiKey":"sk-..."}` and smuggle a + * credential through a type that has no field for it. Rejecting keeps the wire shape exactly the + * declared shape. + */ +export function parsePrincipal(input: unknown): Principal { + if (typeof input !== "object" || input === null || Array.isArray(input)) { + throw new ProtocolError("principal must be a JSON object"); + } + const raw = input as Record; + + const allowed = new Set([ + "id", + "kind", + "centerId", + "displayName", + "createdAt", + "revokedAt", + ]); + for (const key of Object.keys(raw)) { + if (!allowed.has(key)) { + throw new ProtocolError( + `unknown field "${key}" on principal. Principals are a closed shape: an unrecognised ` + + `field is either a typo or something that does not belong in a principal at all.`, + ); + } + } + + const kind = raw.kind; + if (typeof kind !== "string" || !KINDS.includes(kind as PrincipalKind)) { + throw new ProtocolError(`kind must be one of ${KINDS.join(" | ")}`); + } + if (raw.revokedAt !== null && raw.revokedAt !== undefined) { + assertFiniteNumber(raw.revokedAt, "revokedAt"); + } + + return { + id: assertNonEmptyString(raw.id, "id"), + kind: kind as PrincipalKind, + centerId: assertNonEmptyString(raw.centerId, "centerId"), + displayName: assertNonEmptyString(raw.displayName, "displayName"), + createdAt: assertFiniteNumber(raw.createdAt, "createdAt"), + revokedAt: (raw.revokedAt ?? null) as number | null, + }; +} + +export function serializePrincipal(principal: Principal): string { + // Round-trip through the validator so serialization cannot emit a shape parse() would reject. + return JSON.stringify(parsePrincipal(principal)); +} + +/** True when the principal has been revoked. Revocation is soft, so this is a fact, not a state. */ +export function isRevoked(principal: Principal, now: number): boolean { + return principal.revokedAt !== null && principal.revokedAt <= now; +} + +export interface CreatePrincipalOptions { + readonly id: string; + readonly kind: PrincipalKind; + readonly centerId: string; + readonly displayName: string; + readonly now: number; +} + +/** Construct a live principal. There is no parameter through which a secret could be passed. */ +export function createPrincipal(options: CreatePrincipalOptions): Principal { + return parsePrincipal({ + id: options.id, + kind: options.kind, + centerId: options.centerId, + displayName: options.displayName, + createdAt: options.now, + revokedAt: null, + }); +} diff --git a/packages/protocol/src/sync.ts b/packages/protocol/src/sync.ts new file mode 100644 index 0000000..c6c15ef --- /dev/null +++ b/packages/protocol/src/sync.ts @@ -0,0 +1,174 @@ +/** + * The host↔plane seam: tenant resolution, and the outbox-then-settle pattern. + * + * Both are the parts of Phase 2 that are *logic* rather than *deployment*, which is why they can + * be written and proved here. What genuinely needs Cloudflare is the Durable Object that will + * host them — single-threaded execution per tenant is the isolation boundary, and that is a + * property of a runtime we do not have yet. So this module is deliberately storage-agnostic: it + * takes whatever store you hand it, and a DO adapter becomes a thin layer later. + * + * ── Tenant resolution is the security boundary ────────────────────────────────────────────── + * `DATA-MODEL.md` §1: "Resolved once, at the edge, from a verified token. **Never** from a + * header the client can set freely. Unresolved tenant → reject before touching a DO, not inside + * one." The last clause is load-bearing: resolving inside the DO means the DO has already been + * addressed by an id the attacker chose. + * + * ── Outbox-then-settle, not distributed transactions ──────────────────────────────────────── + * The pattern is borrowed from antiproton's *shape* only (reimplemented per D-005 — it has zero + * tests, so nothing was copied). The rule that makes it correct: the outbox row and the state + * change it describes are written in the *same* durable step, then a separate settle publishes. A + * crash between them leaves a pending row, never a lost one — a redelivery problem every + * consumer already handles, rather than a lost write that nothing does. + */ + +/** A tenant, resolved at the edge. Never inferred from a client-supplied header. */ +export interface Tenant { + readonly tenantId: string; + /** The Durable Object that owns this tenant's data. Resolved, never client-chosen. */ + readonly durableObjectId: string; +} + +export class UnresolvedTenantError extends Error { + constructor(readonly detail: string) { + super( + `tenant unresolved: ${detail}. Rejecting at the edge — resolving inside a Durable Object ` + + `would mean the object had already been addressed by an id the client chose.`, + ); + this.name = "UnresolvedTenantError"; + } +} + +export interface TenantResolver { + /** Turn a verified token into a tenant, or throw. Must not consult a client header. */ + resolve(verifiedToken: string): Tenant; +} + +/** + * An in-memory tenant table, standing in for the identity service until one exists. + * + * Deliberately strict: an unknown token throws rather than returning a default, because a + * default tenant is a silent cross-tenant read. + */ +export class StaticTenantTable implements TenantResolver { + readonly #byToken = new Map(); + + constructor(tenants: readonly Tenant[] = []) { + for (const t of tenants) + this.#byToken.set(`${t.tenantId}:${t.durableObjectId}`, t); + } + + add(tenant: Tenant, token: string): void { + this.#byToken.set(token, tenant); + } + + resolve(verifiedToken: string): Tenant { + if (typeof verifiedToken !== "string" || verifiedToken.length === 0) { + throw new UnresolvedTenantError("no token presented"); + } + const tenant = this.#byToken.get(verifiedToken); + if (!tenant) { + throw new UnresolvedTenantError( + `token ${verifiedToken} is not a known tenant`, + ); + } + return tenant; + } +} + +/** A durable row the caller must settle once it is safe to publish. */ +export interface OutboxEntry { + readonly id: string; + readonly tenantId: string; + readonly payload: T; + readonly createdAt: number; + settledAt: number | null; +} + +/** + * Outbox, with idempotent settle. + * + * The property that matters: a pending entry is never silently dropped. If publishing throws or + * the process dies, the entry stays pending and is retried — redelivery, which every consumer + * already handles, rather than a lost write, which nothing does. + */ +export class Outbox { + readonly #entries = new Map(); + + /** Write the outbox row in the same durable step as the state it describes. */ + enqueue(entry: Omit, "settledAt">): void { + if (this.#entries.has(entry.id)) { + // Re-enqueuing the same id is a retry, not a duplicate. Silently replacing it would let a + // concurrent write clobber a row another worker is still settling. + return; + } + this.#entries.set(entry.id, { ...entry, settledAt: null }); + } + + /** + * Settle one entry. Retrying after a partial publish is safe because the handler is expected to + * be idempotent, and a double-settle is a no-op rather than a second publish. + */ + settle(id: string, at: number): boolean { + const entry = this.#entries.get(id); + if (!entry) return false; + if (entry.settledAt !== null) return false; // already settled + this.#entries.set(id, { ...entry, settledAt: at }); + return true; + } + + /** Entries still awaiting publish. After a crash, this is the retry list. */ + pending(): readonly OutboxEntry[] { + return [...this.#entries.values()].filter((e) => e.settledAt === null); + } + + get size(): number { + return this.#entries.size; + } +} + +/** A plane cursor — the thing the host syncs, instead of tokens. */ +export interface Cursor { + readonly sessionId: string; + /** Highest `seq` the plane has durably accepted for this stream. */ + readonly afterSeq: number; +} + +/** + * Merge a host's records into the plane's view, given where it already is. + * + * `PROTOCOL.md` §4 is explicit that everything except `approvals` is **at-least-once with + * idempotent consumers**, and this is that consumer. The guarantee it provides, and no more: + * after a merge the plane's `afterSeq` equals the highest `seq` it holds, and replaying the same + * batch changes nothing. That is what makes an offline host safe to reconnect — and it is + * deliberately *not* exactly-once, which is the claim the docs reserve for `approvals`. + */ +export class CursorMerger { + readonly #cursors = new Map(); + + /** Records are identified by `sessionId` + `seq`. Returns how many were newly accepted. */ + merge( + sessionId: string, + records: readonly { readonly seq: number }[], + ): number { + if (records.length === 0) return 0; + const sorted = [...records].sort((a, b) => a.seq - b.seq); + let current = this.#cursors.get(sessionId) ?? -1; + let accepted = 0; + for (const record of sorted) { + if (record.seq <= current) continue; // already held — a replay, not new data + current = record.seq; + accepted += 1; + } + this.#cursors.set(sessionId, current); + return accepted; + } + + afterSeq(sessionId: string): number { + return this.#cursors.get(sessionId) ?? -1; + } + + /** What the host should send next: everything after the plane's cursor. */ + nextCursor(sessionId: string): Cursor { + return { sessionId, afterSeq: this.afterSeq(sessionId) }; + } +} diff --git a/packages/protocol/src/version.ts b/packages/protocol/src/version.ts new file mode 100644 index 0000000..32b36b4 --- /dev/null +++ b/packages/protocol/src/version.ts @@ -0,0 +1,177 @@ +/** + * `radius/v1` — the envelope, and the compatibility rules that make a promise you can keep. + * + * `ROADMAP.md` calls this "the real moat": once developers build agents against `radius/v1`, the + * platform gets stickier than any feature set. That only holds if versioning is *enforced* rather + * than promised, so the rules from `PROTOCOL.md` §8 are implemented here as a checker rather than + * left as prose. + * + * ── The rules, and the one that surprises people ─────────────────────────────────────────── + * new optional field ............ minor + * new record kind ............... minor + * removal, rename, semantic ..... MAJOR + * **tightening a default ....... MAJOR** + * + * That last one is load-bearing and the reason this file exists. Turning a sandbox on by default + * is a *breaking* change for a host built against v1 defaults, even though no field changed — a + * host that silently loses its sandbox must fail loudly, not drift. A tool that only diffs field + * names cannot see that, so `classifyRelease` works on **semantic** facts, not shapes. + * + * "A host may lag the plane by one minor. Never a major." — enforced in `assertCompatible`. + */ + +import { ProtocolError } from "./principal.ts"; + +// Re-exported so a consumer of the version surface catches the malformed-version case without +// also reaching into the principal module. Same as `lease.ts`. +export { ProtocolError }; + +export const PROTOCOL_VERSION = "radius/v1"; +export const PROTOCOL_MAJOR = 1; +export const PROTOCOL_MINOR = 0; + +export interface ParsedVersion { + readonly major: number; + readonly minor: number; +} + +/** Parse `radius/vN` or `radius/vN.M`. Rejects anything else rather than guessing. */ +export function parseVersion(raw: string): ParsedVersion { + const match = /^radius\/v(\d+)(?:\.(\d+))?$/.exec(raw); + if (!match?.[1]) { + throw new ProtocolError( + `unrecognised protocol version "${raw}". Expected ${PROTOCOL_VERSION} or radius/vN.M.`, + ); + } + return { major: Number(match[1]), minor: Number(match[2] ?? 0) }; +} + +export function formatVersion({ major, minor }: ParsedVersion): string { + return `radius/v${major}.${minor}`; +} + +/** The wire envelope. `v` is required — `PROTOCOL.md` §2 says so explicitly. */ +export interface Envelope { + readonly v: string; + readonly requestId: string; + readonly payload: T; +} + +export class UnsupportedVersionError extends Error { + constructor( + readonly theirs: string, + readonly ours: string, + ) { + super(`unsupported protocol version ${theirs}; this build speaks ${ours}`); + this.name = "UnsupportedVersionError"; + } +} + +/** + * Can a host at `host` talk to a plane at `plane`? + * + * `PROTOCOL.md` §8: "A host may lag the plane by one minor. Never a major." So a newer *minor* on + * the host is tolerated, a major mismatch is not, and more than one minor of lag is not. Being + * strict here is the point — a silently-mismatched pair is how a "backwards compatible" API + * stops being one. + */ +export function assertCompatible(plane: string, host: string): void { + const p = parseVersion(plane); + const h = parseVersion(host); + if (p.major !== h.major) { + throw new UnsupportedVersionError(host, plane); + } + if (p.minor - h.minor > 1) { + throw new ProtocolError( + `host is ${p.minor - h.minor} minors behind the plane, and only one minor of lag is ` + + `supported. Upgrade the host rather than assuming compatibility.`, + ); + } +} + +/** A semantic fact about a release — deliberately not a field-level diff. */ +export interface ReleaseNote { + readonly version: string; + readonly newOptionalFields?: readonly string[]; + readonly newRecordKinds?: readonly string[]; + readonly removedOrRenamed?: readonly string[]; + readonly semanticChanges?: readonly string[]; + /** A default that got STRICTER, e.g. sandbox on. This is a major, by rule. */ + readonly tightenedDefaults?: readonly string[]; +} + +export type ChangeClass = "minor" | "major" | "none"; + +/** + * Classify one release. Pure, so the policy is tested rather than trusted. + * + * The two interesting outputs are the ones that are easy to get wrong: a tightened default is + * **major** even with nothing added, and an *empty* release is neither minor nor major. + */ +export function classifyRelease(note: ReleaseNote): ChangeClass { + if ( + (note.removedOrRenamed?.length ?? 0) > 0 || + (note.semanticChanges?.length ?? 0) > 0 || + (note.tightenedDefaults?.length ?? 0) > 0 + ) { + return "major"; + } + if ( + (note.newOptionalFields?.length ?? 0) > 0 || + (note.newRecordKinds?.length ?? 0) > 0 + ) { + return "minor"; + } + return "none"; +} + +export class BreakingReleaseError extends Error { + constructor( + readonly version: string, + readonly reasons: readonly string[], + ) { + super( + `release ${version} is breaking: ${reasons.join("; ")}. A tightened default counts as ` + + `breaking even when no field changed — a host that silently loses its sandbox must fail ` + + `loudly, not drift. Bump the major version.`, + ); + this.name = "BreakingReleaseError"; + } +} + +/** + * Assert a release does the right thing with its version number. + * + * Fails in both directions, deliberately: a breaking change shipped as a minor, *and* a minor + * shipped as a major. The second looks harmless and is not — it forces every host to upgrade for + * an additive change, which is exactly how a "stable API" becomes a moving target. + */ +export function assertVersionBump(previous: string, note: ReleaseNote): void { + const from = parseVersion(previous); + const to = parseVersion(note.version); + const kind = classifyRelease(note); + + if (to.major < from.major) { + throw new ProtocolError( + `version went backwards: ${previous} → ${note.version}`, + ); + } + + if (kind === "major" && to.major === from.major) { + throw new BreakingReleaseError( + note.version, + [ + ...(note.removedOrRenamed ?? []).map((f) => `removed/renamed ${f}`), + ...(note.semanticChanges ?? []).map((c) => `semantic: ${c}`), + ...(note.tightenedDefaults ?? []).map((d) => `tightened default: ${d}`), + ].slice(0, 3), + ); + } + + if (kind === "minor" && to.major > from.major) { + throw new ProtocolError( + `${note.version} is an additive change but bumped the MAJOR version. That forces every ` + + `host to upgrade for an optional field; it is a minor.`, + ); + } +} diff --git a/packages/protocol/test/approvals.test.ts b/packages/protocol/test/approvals.test.ts new file mode 100644 index 0000000..8c3a6c2 --- /dev/null +++ b/packages/protocol/test/approvals.test.ts @@ -0,0 +1,256 @@ +/** + * Approvals (D-013) and hard spend caps (the Phase 4 gate). + * + * Both are easy to claim and easy to quietly break, so the tests assert the *mechanism* — + * exactly-once by construction, and `committed + reserved <= cap` — rather than an outcome that + * happens to imply it. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + ApprovalsStore, + ApprovalConflictError, + MissingDenialReasonError, + NonHumanDecisionError, + type ApprovalDecision, + type ApprovalRequest, +} from "../src/approvals.ts"; +import { BudgetExceededError, CostLedger } from "../src/budget.ts"; + +const request = (id = "req_1"): ApprovalRequest => ({ + requestId: id, + principalId: "pr_1", + capability: "net:egress", + scope: { hosts: ["api.example.com"] }, + requestedAt: 1000, +}); + +const decision = (over: Partial = {}): ApprovalDecision => ({ + requestId: "req_1", + outcome: "approved", + decidedBy: "pr_human", + decidedByKind: "human", + decidedAt: 1001, + reason: null, + ...over, +}); + +describe("Approvals — exactly-once, by construction (D-013)", () => { + test("a first decision is recorded", () => { + const store = new ApprovalsStore(); + assert.equal(store.submit(request(), decision()).kind, "recorded"); + assert.equal(store.size, 1); + }); + + test("a client retry returns the original and does not add a row", () => { + // This IS the exactly-once claim. Not "we are careful" — there is one slot keyed by + // requestId, so a retry physically cannot produce a second row. + const store = new ApprovalsStore(); + const first = store.submit(request(), decision({ decidedAt: 1001 })); + const second = store.submit(request(), decision({ decidedAt: 9999 })); + + assert.equal(second.kind, "replayed"); + assert.equal(store.size, 1, "a retry must not grow the table"); + assert.equal( + store.get("req_1")?.decidedAt, + 1001, + "the original stands unchanged", + ); + assert.equal(second.decision.decidedAt, first.decision.decidedAt); + }); + + test("many concurrent retries still produce exactly one row", () => { + const store = new ApprovalsStore(); + for (let i = 0; i < 50; i++) + store.submit(request(), decision({ decidedAt: 1000 + i })); + assert.equal(store.size, 1); + }); + + test("a replay with different content is refused, not overwritten", () => { + // The dangerous case: same requestId, different scope. Overwriting would let an agent + // re-approve a decision it was never granted. + const store = new ApprovalsStore(); + store.submit(request(), decision()); + assert.throws( + () => + store.submit( + { ...request(), scope: { hosts: ["evil.example.com"] } }, + decision(), + ), + ApprovalConflictError, + ); + assert.deepEqual( + store.get("req_1")?.outcome, + "approved", + "the original is intact", + ); + }); + + test("a different capability under the same requestId is a conflict", () => { + const store = new ApprovalsStore(); + store.submit(request(), decision()); + assert.throws( + () => store.submit({ ...request(), capability: "fs:write" }, decision()), + ApprovalConflictError, + ); + }); + + test("an agent cannot decide its own escalation (D-014)", () => { + const store = new ApprovalsStore(); + assert.throws( + () => + store.submit( + request(), + decision({ decidedByKind: "agent", decidedBy: "pr_1" }), + ), + NonHumanDecisionError, + ); + assert.equal(store.size, 0, "a refused decision leaves no row"); + }); + + test("a service cannot decide either", () => { + const store = new ApprovalsStore(); + assert.throws( + () => store.submit(request(), decision({ decidedByKind: "service" })), + NonHumanDecisionError, + ); + }); + + test("a denial with no reason is refused", () => { + // A refusal an operator cannot explain is not a usable refusal. + const store = new ApprovalsStore(); + assert.throws( + () => + store.submit(request(), decision({ outcome: "denied", reason: null })), + MissingDenialReasonError, + ); + }); + + test("a denial with a reason is recorded", () => { + const store = new ApprovalsStore(); + store.submit( + request(), + decision({ outcome: "denied", reason: "unexpected egress" }), + ); + assert.equal(store.get("req_1")?.reason, "unexpected egress"); + }); +}); + +describe("CostLedger — a hard cap under fault injection", () => { + const ledger = (cap = 100) => + new CostLedger({ capMinor: cap, currency: "usd" }); + + test("a turn within the cap is allowed and settles", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 60, 1); + l.settle(r.id, 45); + assert.equal(l.committedFor("pr_1"), 45); + assert.equal(l.available("pr_1"), 55, "unused headroom returns"); + }); + + test("spending past the cap is refused", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 100, 1); + l.settle(r.id, 100); + assert.throws(() => l.reserve("pr_1", 1, 2), BudgetExceededError); + }); + + test("two concurrent reservations cannot together cross the cap", () => { + // The race the gate warns about. Checking only `committed` would let both through. + const l = ledger(100); + const a = l.reserve("pr_1", 60, 1); + assert.throws( + () => l.reserve("pr_1", 60, 2), + BudgetExceededError, + "the second reservation sees the first one's headroom", + ); + l.settle(a.id, 60); + }); + + test("a retry that settles twice is charged once", () => { + // "A provider that fails after accepting has already billed it" — the failure mode that + // makes naive caps overshoot. A repeated settlement must not double-charge. + const l = ledger(100); + const r = l.reserve("pr_1", 50, 1); + l.settle(r.id, 40); + l.settle(r.id, 40); + assert.equal( + l.committedFor("pr_1"), + 40, + "a repeated settlement is a no-op", + ); + }); + + test("a partial result returns the unused headroom", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 80, 1); + l.settle(r.id, 10); + assert.equal( + l.available("pr_1"), + 90, + "70 of reservation went back to the budget", + ); + }); + + test("an actual cost above the reservation is refused, not absorbed", () => { + // If the real cost exceeds what we reserved, the cap has already been crossed. Saying so is + // the honest outcome; quietly over-committing is how a "hard cap" becomes a soft one. + const l = ledger(100); + const r = l.reserve("pr_1", 10, 1); + assert.throws(() => l.settle(r.id, 50), BudgetExceededError); + assert.equal(l.committedFor("pr_1"), 0, "nothing was committed"); + }); + + test("a crash between reserve and send does not leak headroom", () => { + // Without release(), a reservation would sit forever and the agent would stop working with + // budget still available — the opposite failure, and just as bad. + const l = ledger(100); + const a = l.reserve("pr_1", 90, 1); + assert.equal(l.available("pr_1"), 10); + l.release(a.id); + assert.equal(l.available("pr_1"), 100, "the headroom came back"); + assert.equal(l.openReservations, 0); + }); + + test("committed + reserved never exceeds the cap, across a mixed run", () => { + // The invariant itself, asserted directly rather than inferred from any single outcome. + const l = ledger(100); + for (let i = 0; i < 20; i++) { + const bound = 10 + ((i * 7) % 40); + try { + const r = l.reserve("pr_1", bound, i); + l.settle(r.id, Math.floor(bound / 2)); + } catch { + /* a refusal is a valid outcome here */ + } + assert.ok( + l.committedFor("pr_1") + l.reservedFor("pr_1") <= l.capMinor, + "committed + reserved must never exceed the cap", + ); + } + }); + + test("budgets are per principal", () => { + const l = ledger(100); + const r = l.reserve("pr_1", 100, 1); + l.settle(r.id, 100); + assert.doesNotThrow( + () => l.reserve("pr_2", 100, 2), + "pr_2 has its own budget", + ); + }); + + test("a negative or fractional amount is refused", () => { + const l = ledger(100); + assert.throws(() => l.reserve("pr_1", -1, 1), /non-negative/); + assert.throws(() => l.reserve("pr_1", 1.5, 1), /non-negative/); + }); + + test("an unknown reservation cannot be settled or released", () => { + const l = ledger(100); + assert.throws(() => l.settle("nope", 1), /no such reservation/); + assert.throws(() => l.release("nope"), /no such reservation/); + }); +}); diff --git a/packages/protocol/test/lease.test.ts b/packages/protocol/test/lease.test.ts new file mode 100644 index 0000000..6a81ccf --- /dev/null +++ b/packages/protocol/test/lease.test.ts @@ -0,0 +1,313 @@ +/** + * Lease semantics — the properties `MILESTONES.md` 1.4 lists, each mapped to the invariant it + * now enforces in code rather than in prose. + * + * I3 an expired lease is refused + * I5 a stale lease epoch is rejected on every write + * I14 a monotonic lease is not extended by wall-clock rollback + * + * Time is injected, never slept on. These are logic tests about an authority's clock, and a test + * that waits 60 seconds to observe an hour-long TTL has proved nothing except that it is slow. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + LeaseExpiredError, + LeaseHeldError, + LeaseManager, + MonotonicClock, + StaleEpochError, + type MonotonicSource, +} from "../src/lease.ts"; + +/** A monotonic source the test drives by hand. */ +class FakeMonotonic implements MonotonicSource { + #t = 0; + now(): number { + return this.#t; + } + advance(ms: number): void { + this.#t += ms; + } +} + +function setup(ttlMs = 1000) { + const source = new FakeMonotonic(); + // A wall clock the test can move BACKWARDS, which is the whole attack D-008 Q4 describes. + let wall = 1_700_000_000_000; + const clock = new MonotonicClock({ source, wallNow: () => wall }); + const leases = new LeaseManager({ clock, ttlMs }); + return { + source, + leases, + clock, + setWall: (v: number) => { + wall = v; + }, + }; +} + +describe("LeaseManager — exclusivity", () => { + test("one controller per resource; a second acquire is refused", () => { + // MILESTONES 1.4: "One controller per session; leases are the exclusive claim". + const { leases } = setup(); + leases.acquire("session:a", "host-1"); + assert.throws(() => leases.acquire("session:a", "host-2"), LeaseHeldError); + }); + + test("the epoch increments on every acquisition", () => { + const { leases, source } = setup(100); + assert.equal(leases.acquire("session:a", "host-1").epoch, 1); + source.advance(200); // lease 1 expires + assert.equal( + leases.acquire("session:a", "host-2").epoch, + 2, + "takeover must bump the epoch", + ); + }); + + test("a dead holder can be taken over once its lease expires", () => { + const { leases, source } = setup(100); + leases.acquire("session:a", "host-dead"); + source.advance(101); + const taken = leases.acquire("session:a", "host-live"); + assert.equal(taken.holderId, "host-live"); + }); +}); + +describe("LeaseManager — I5: fencing rejects a partitioned holder", () => { + test("a stale epoch is rejected on every write, forever", () => { + // This is the split-brain test. The partitioned holder is NOT aware it lost the lease — from + // its side everything looks fine. Only the epoch check stops it. Reasoning about why this + // works proves nothing; the assertion does. + const { leases, source } = setup(100); + const stale = leases.acquire("session:a", "host-1"); + + source.advance(101); + const live = leases.acquire("session:a", "host-2"); + + // The new holder can write. + assert.equal( + leases.assertWritable("session:a", "host-2", live.epoch).epoch, + 2, + ); + + // The partitioned holder still believes it owns the lease, and is refused anyway. + assert.throws( + () => leases.assertWritable("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + // ...and stays refused, however many times it retries. + for (let i = 0; i < 5; i++) { + assert.throws( + () => leases.assertWritable("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + } + }); + + test("a stale holder cannot renew, release, or extend the new holder's lease", () => { + const { leases, source } = setup(100); + const stale = leases.acquire("session:a", "host-1"); + source.advance(101); + const live = leases.acquire("session:a", "host-2"); + + assert.ok( + live.epoch > stale.epoch, + "a takeover strictly raises the epoch — that is the fencing token", + ); + assert.throws( + () => leases.renew("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + + describe("LeaseManager — I3: expiry", () => { + test("an expired lease is refused on write", () => { + const { leases, source } = setup(100); + const lease = leases.acquire("session:a", "host-1"); + assert.ok(leases.assertWritable("session:a", "host-1", lease.epoch)); + source.advance(101); + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + ); + }); + + test("an expired lease does not report as current", () => { + // Reporting an expired lease as live is how a second controller convinces itself it may + // write. The authority's view must say "nothing here". + const { leases, source } = setup(100); + leases.acquire("session:a", "host-1"); + assert.ok(leases.current("session:a"), "live while held"); + source.advance(101); + assert.equal(leases.current("session:a"), null, "gone once expired"); + }); + }); + + describe("LeaseManager — I14: clock rollback cannot extend a lease", () => { + test("moving the wall clock BACKWARDS does not revive an expired lease", () => { + // The attack D-008 Q4 describes, executed: the user owns the wall clock. Expiry is computed + // from the monotonic source, which they cannot reach, so the lease stays dead. + const { leases, source, setWall } = setup(1000); + const lease = leases.acquire("session:a", "host-1"); + + source.advance(1001); // monotonic time passes; lease expires + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + ); + + // Now rewind the wall clock a year. A wall-clock lease would spring back to life here. + setWall(1_600_000_000_000); + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + "a rewound wall clock must not resurrect an expired lease", + ); + }); + + test("a lease taken AFTER the rollback is still bounded by monotonic time", () => { + const { leases, source, setWall } = setup(1000); + setWall(1_600_000_000_000); + const lease = leases.acquire("session:a", "host-1"); + source.advance(1001); + assert.throws( + () => leases.assertWritable("session:a", "host-1", lease.epoch), + LeaseExpiredError, + ); + }); + }); + + describe("MonotonicClock — survives a restart", () => { + test("monotonic time is continuous across a restart", () => { + // D-008: "Restarting the process resets the monotonic clock, so the reconciliation must be + // re-derived from persisted state." A live lease must NOT expire merely because the host + // restarted. + const source = new FakeMonotonic(); + const first = new MonotonicClock({ source }); + + source.advance(30_000); // 30s elapse in process one + const persisted = first.anchor(); + assert.ok( + persisted.monotonicAt >= 30_000, + "time advanced before the restart", + ); + + // Process two: a fresh clock whose source has reset to zero, as hrtime does. + const restarted = new MonotonicClock({ + source: { now: () => 0 }, + anchor: persisted, + }); + assert.equal( + restarted.now(), + persisted.monotonicAt, + "resumes exactly where it left off", + ); + }); + + test("a lease still live before a restart is still live after it", () => { + // The property a restart must not break: a 10s lease with 1s consumed is not suddenly dead + // because the process bounced. Without the anchor being restored, hrtime resets to ~0 and + // every live lease looks centuries old. + const source = new FakeMonotonic(); + const before = new LeaseManager({ + clock: new MonotonicClock({ source }), + ttlMs: 10_000, + }); + const lease = before.acquire("session:a", "host-1"); + source.advance(1_000); + + // Process two: a fresh source that has reset, plus the persisted state. + const restartedSource = new FakeMonotonic(); + const after = new LeaseManager({ + clock: new MonotonicClock({ source: restartedSource }), + ttlMs: 10_000, + }); + after.restore( + before.entries(), + before.entries().length + ? clockAnchorOf(before) + : { monotonicAt: 0, wallAt: 0 }, + ); + + assert.ok( + after.current("session:a"), + "a lease that was live before the restart is still live after it", + ); + assert.equal( + after.current("session:a")?.epoch, + lease.epoch, + "the epoch is preserved", + ); + }); + + test("a lease whose window elapsed BEFORE the restart is expired after it", () => { + // D-008 Q4's explicit case: "if the persisted state shows the monotonic window has already + // elapsed, the lease is expired regardless of the wall clock." Here the wall clock will claim + // the lease is young. Restore must not believe it. + const source = new FakeMonotonic(); + const before = new LeaseManager({ + clock: new MonotonicClock({ source }), + ttlMs: 5_000, + }); + before.acquire("session:a", "host-1"); + const persisted = before.entries()[0]; + assert.ok(persisted); + + source.advance(6_000); // it expires while the process is alive + const anchor = { monotonicAt: persisted.expiresAt, wallAt: 0 }; + + const after = new LeaseManager({ + clock: new MonotonicClock({ source: { now: () => 0 } }), + ttlMs: 5_000, + }); + after.restore([persisted], anchor); + + assert.equal( + after.current("session:a"), + null, + "a lease whose window elapsed is dropped on restart, not revived", + ); + }); + }); + + /** The anchor a live LeaseManager would persist alongside its leases. */ + function clockAnchorOf(manager: LeaseManager): { + monotonicAt: number; + wallAt: number; + } { + const leases = manager.entries(); + const earliest = leases.reduce( + (min, l) => Math.min(min, l.acquiredAt), + Infinity, + ); + return { + monotonicAt: Number.isFinite(earliest) ? earliest : 0, + wallAt: 0, + }; + } + + assert.throws( + () => leases.release("session:a", "host-1", stale.epoch), + StaleEpochError, + ); + assert.equal( + leases.current("session:a")?.epoch, + 2, + "the live lease is untouched", + ); + assert.equal(leases.current("session:a")?.holderId, "host-2"); + }); + + test("a holder cannot write under another holder's epoch", () => { + const { leases } = setup(); + const lease = leases.acquire("session:a", "host-1"); + assert.throws( + () => leases.assertWritable("session:a", "host-2", lease.epoch), + LeaseHeldError, + "a correct epoch on the wrong holder is still refused", + ); + }); +}); diff --git a/packages/protocol/test/principal.test.ts b/packages/protocol/test/principal.test.ts new file mode 100644 index 0000000..f198ea4 --- /dev/null +++ b/packages/protocol/test/principal.test.ts @@ -0,0 +1,191 @@ +/** + * Principal model and the secret scan. + * + * The scan is the part that matters. `DATA-MODEL.md` §4 says enforcement is "invariant I1 (no + * secret-shaped value in any durable field) plus a test that scans for it. **Not a code-review + * convention.**" This file is that test. A type without a secret field is a strong claim; a type + * plus a test that would fail if a secret-shaped string ever appeared is a claim you can rely on. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + ProtocolError, + createPrincipal, + isRevoked, + parsePrincipal, + serializePrincipal, + type CapabilityGrant, +} from "../src/principal.ts"; + +const live = () => + createPrincipal({ + id: "pr_01", + kind: "agent", + centerId: "task_42", + displayName: "ci-runner", + now: 1_700_000_000_000, + }); + +/** + * A copy of `value` with one key removed. + * + * Built by filtering rather than `delete` on a computed key: deleting a property the key set is + * derived from is the pattern that lets a fixture drift away from the type it is exercising, and + * a lint rule rightly refuses it. + */ +function omit(value: object, key: string): Record { + return Object.fromEntries(Object.entries(value).filter(([k]) => k !== key)); +} + +describe("Principal — round trip", () => { + test("survives serialize → parse unchanged", () => { + const p = live(); + assert.deepEqual(parsePrincipal(JSON.parse(serializePrincipal(p))), p); + }); + + test("all three kinds round-trip", () => { + for (const kind of ["human", "agent", "service"] as const) { + const p = createPrincipal({ + id: "pr_x", + kind, + centerId: "c", + displayName: "d", + now: 1, + }); + assert.equal( + parsePrincipal(JSON.parse(serializePrincipal(p))).kind, + kind, + ); + } + }); + + test("revocation is soft and round-trips", () => { + const p = { ...live(), revokedAt: 1_700_000_100_000 }; + assert.equal( + parsePrincipal(JSON.parse(serializePrincipal(p))).revokedAt, + p.revokedAt, + ); + assert.equal(isRevoked(p, 1_700_000_100_000), true); + assert.equal( + isRevoked(p, 1_700_000_000_000), + false, + "not revoked before the timestamp", + ); + }); +}); + +describe("Principal — validation", () => { + test("rejects an unknown kind", () => { + assert.throws( + () => parsePrincipal({ ...live(), kind: "root" }), + ProtocolError, + ); + }); + + test("rejects missing and empty required fields", () => { + for (const field of ["id", "centerId", "displayName"]) { + assert.throws( + () => parsePrincipal(omit(live(), field)), + ProtocolError, + `missing ${field}`, + ); + assert.throws( + () => parsePrincipal({ ...omit(live(), field), [field]: "" }), + ProtocolError, + `empty ${field}`, + ); + } + }); + + test("rejects a non-finite createdAt", () => { + assert.throws( + () => parsePrincipal({ ...live(), createdAt: Number.NaN }), + ProtocolError, + ); + }); + + test("rejects a non-object", () => { + for (const bad of [null, "x", 42, []]) { + assert.throws(() => parsePrincipal(bad), ProtocolError); + } + }); +}); + +describe("Principal — the deliberate omission, enforced", () => { + test("a secret cannot ride in as an unknown field", () => { + // The closed-shape rule is what stops a permissive parser from accepting + // `{"id":..., "apiKey":"sk-..."}` — smuggling a credential through a type with no field for it. + assert.throws( + () => + parsePrincipal({ ...live(), apiKey: "sk-ant-api03-REAL-LOOKING-KEY" }), + ProtocolError, + ); + assert.throws( + () => parsePrincipal({ ...live(), token: "ghp_abc123" }), + ProtocolError, + ); + }); + + test("a serialized principal never contains a secret-shaped value", () => { + const p = live(); + const wire = serializePrincipal(p); + for (const pattern of SECRET_PATTERNS) { + assert.ok( + !pattern.test(wire), + `serialized principal must not contain a secret-shaped value (${pattern})`, + ); + } + }); + + test("a serialized grant never contains a secret-shaped value", () => { + // The scope object is attacker-supplied JSON, so it is the realistic leak path. A grant is + // not a Principal, so the closed-shape rule does not cover it — hence this test. + const grant: CapabilityGrant = { + capability: "repo:read", + scope: { repos: ["a", "b"] }, + grantedBy: "pr_human", + grantedAt: 1, + expiresAt: 2, + monotonicBound: 2, + approvalId: "ap_1", + }; + const wire = JSON.stringify(grant); + for (const pattern of SECRET_PATTERNS) { + assert.ok( + !pattern.test(wire), + `serialized grant must not contain a secret (${pattern})`, + ); + } + }); + + test("a scope carrying a secret is detected if it ever reaches the wire", () => { + // A positive control for the scan itself. Without this, a scanner that matched nothing would + // pass every other test in this file and look like a green suite. + const leaky = JSON.stringify({ + capability: "repo:read", + scope: { note: "AKIAIOSFODNN7EXAMPLE" }, + }); + assert.ok( + SECRET_PATTERNS.some((p) => p.test(leaky)), + "the scanner must actually detect a planted secret, or it is not a check", + ); + }); +}); + +/** + * Shapes worth refusing in any durable structure. Deliberately pattern-based rather than + * value-based: the goal is to catch an accidental credential, not to adjudicate every string. + */ +const SECRET_PATTERNS: readonly RegExp[] = [ + /sk-ant-[A-Za-z0-9_-]{8,}/, // Anthropic + /sk-[A-Za-z0-9]{20,}/, // OpenAI-style + /gh[pousr]_[A-Za-z0-9]{16,}/, // GitHub + /AKIA[0-9A-Z]{16}/, // AWS access key id + /xox[baprs]-[A-Za-z0-9-]{10,}/, // Slack + /AIza[0-9A-Za-z_-]{20,}/, // Google + /-----BEGIN [A-Z ]*PRIVATE KEY-----/, + /eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\./, // JWT + /"?(?:api[_-]?key|secret|password|passwd|bearer|authorization)"?\s*[:=]\s*"[^"]{8,}"/i, +]; diff --git a/packages/protocol/test/sync.test.ts b/packages/protocol/test/sync.test.ts new file mode 100644 index 0000000..fb10cc1 --- /dev/null +++ b/packages/protocol/test/sync.test.ts @@ -0,0 +1,195 @@ +/** + * Phase 2 logic: tenant resolution, outbox-then-settle, and cursor merge. + * + * This is the part of the control plane that is *logic* rather than *deployment*. It is fully + * tested here; what is NOT tested — and cannot be without Cloudflare — is the Durable Object + * that will host it, since single-threaded-per-tenant execution is the actual isolation boundary. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + CursorMerger, + Outbox, + StaticTenantTable, + UnresolvedTenantError, +} from "../src/sync.ts"; + +const tenant = { tenantId: "t_a", durableObjectId: "do_1" }; + +describe("Tenant resolution — the edge, never a client header", () => { + test("a known token resolves to its tenant and durable object", () => { + const table = new StaticTenantTable(); + table.add(tenant, "token-a"); + assert.deepEqual(table.resolve("token-a"), tenant); + }); + + test("an unknown token is rejected, never defaulted", () => { + // A default tenant would be a silent cross-tenant read — the exact thing §1 forbids. + const table = new StaticTenantTable(); + table.add(tenant, "token-a"); + assert.throws(() => table.resolve("token-b"), UnresolvedTenantError); + }); + + test("an empty or absent token is rejected", () => { + const table = new StaticTenantTable(); + assert.throws(() => table.resolve(""), UnresolvedTenantError); + assert.throws( + () => table.resolve(undefined as unknown as string), + UnresolvedTenantError, + ); + }); + + test("the rejection says it happens at the edge", () => { + // The message is the record of *why* this is checked here and not inside the DO. + const table = new StaticTenantTable(); + assert.throws(() => table.resolve("nope"), /before the object|at the edge/); + }); + + test("two tenants get two distinct durable objects", () => { + const table = new StaticTenantTable(); + table.add({ tenantId: "t_a", durableObjectId: "do_1" }, "token-a"); + table.add({ tenantId: "t_b", durableObjectId: "do_2" }, "token-b"); + assert.notEqual( + table.resolve("token-a").durableObjectId, + table.resolve("token-b").durableObjectId, + ); + }); +}); + +describe("Outbox — a pending row is never lost", () => { + const entry = (id: string) => ({ + id, + tenantId: "t_a", + payload: { n: 1 }, + createdAt: 100, + }); + + test("an enqueued entry is pending until settled", () => { + const box = new Outbox(); + box.enqueue(entry("e1")); + assert.equal(box.pending().length, 1); + assert.equal(box.settle("e1", 200), true); + assert.equal(box.pending().length, 0); + }); + + test("a second settle is a no-op, not a second publish", () => { + const box = new Outbox(); + box.enqueue(entry("e1")); + box.settle("e1", 200); + assert.equal( + box.settle("e1", 300), + false, + "retrying a settle must not republish", + ); + assert.equal(box.size, 1); + }); + + test("re-enqueuing the same id does not duplicate or clobber", () => { + const box = new Outbox(); + box.enqueue(entry("e1")); + box.enqueue({ ...entry("e1"), payload: { n: 999 } }); + assert.equal(box.size, 1); + const kept = box.pending()[0]; + assert.deepEqual( + kept?.payload, + { n: 1 }, + "the original payload is not overwritten", + ); + }); + + test("a crash before settle leaves the entry on the retry list", () => { + // The whole point: redelivery, not loss. + const box = new Outbox(); + box.enqueue(entry("e1")); + assert.equal( + box.pending().length, + 1, + "an unsettled entry survives for retry", + ); + }); + + test("settling an unknown entry reports false rather than throwing", () => { + assert.equal(new Outbox().settle("nope", 1), false); + }); +}); + +describe("CursorMerger — at-least-once, idempotent", () => { + const batch = (from: number, to: number) => + Array.from({ length: to - from + 1 }, (_, i) => ({ seq: from + i })); + + test("a first batch is fully accepted", () => { + const m = new CursorMerger(); + assert.equal(m.merge("s1", batch(0, 4)), 5); + assert.equal(m.afterSeq("s1"), 4); + }); + + test("a replayed batch changes nothing", () => { + // Deliberately NOT exactly-once — PROTOCOL.md reserves that for approvals. This is the + // at-least-once with an idempotent consumer, which is what the stream path is. + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + assert.equal( + m.merge("s1", batch(0, 4)), + 0, + "a full replay accepts nothing new", + ); + assert.equal(m.afterSeq("s1"), 4); + }); + + test("an overlapping batch accepts only the new tail", () => { + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + assert.equal( + m.merge("s1", batch(3, 8)), + 4, + "seq 3 and 4 were already held", + ); + assert.equal(m.afterSeq("s1"), 8); + }); + + test("out-of-order delivery is accepted correctly", () => { + // A host reconnecting may replay in any order; the cursor is by seq, not arrival. + const m = new CursorMerger(); + assert.equal(m.merge("s1", [{ seq: 3 }, { seq: 1 }, { seq: 2 }]), 3); + assert.equal( + m.afterSeq("s1"), + 3, + "the highest seq, regardless of arrival order", + ); + }); + + test("sessions are independent", () => { + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + assert.equal(m.merge("s2", batch(0, 2)), 3, "s2 has its own cursor"); + assert.equal(m.afterSeq("s1"), 4); + }); + + test("an empty batch is a no-op", () => { + const m = new CursorMerger(); + assert.equal(m.merge("s1", []), 0); + assert.equal(m.afterSeq("s1"), -1); + }); + + test("nextCursor tells the host exactly where to resume", () => { + const m = new CursorMerger(); + m.merge("s1", batch(0, 9)); + assert.deepEqual(m.nextCursor("s1"), { sessionId: "s1", afterSeq: 9 }); + }); + + test("resuming from a cursor delivers the remainder with no gap", () => { + // The offline-host loop: sync, go away, come back, resume. + const m = new CursorMerger(); + m.merge("s1", batch(0, 4)); + const { afterSeq } = m.nextCursor("s1"); + const resumed = batch(afterSeq + 1, 9); + assert.equal( + m.merge("s1", resumed), + 5, + "exactly the records after the cursor", + ); + assert.equal(m.afterSeq("s1"), 9); + }); +}); diff --git a/packages/protocol/test/version.test.ts b/packages/protocol/test/version.test.ts new file mode 100644 index 0000000..0828357 --- /dev/null +++ b/packages/protocol/test/version.test.ts @@ -0,0 +1,218 @@ +/** + * `radius/v1` versioning and compatibility. + * + * The tests that matter most are the two that fail in *both* directions — a breaking change + * shipped as a minor, and a minor shipped as a major. The second is the one a careful team gets + * wrong, because it feels safe: it only ever inconveniences people, never breaks them. It is + * also how a "stable API" quietly becomes a moving target. + */ + +import assert from "node:assert/strict"; +import { test, describe } from "node:test"; + +import { + PROTOCOL_VERSION, + BreakingReleaseError, + ProtocolError, + UnsupportedVersionError, + assertCompatible, + assertVersionBump, + classifyRelease, + formatVersion, + parseVersion, +} from "../src/version.ts"; + +describe("version parsing", () => { + test("parses the protocol version and an explicit minor", () => { + assert.deepEqual(parseVersion(PROTOCOL_VERSION), { major: 1, minor: 0 }); + assert.deepEqual(parseVersion("radius/v2.7"), { major: 2, minor: 7 }); + }); + + test("round-trips through format", () => { + assert.equal(formatVersion(parseVersion("radius/v3.4")), "radius/v3.4"); + }); + + test("rejects anything unrecognised rather than guessing", () => { + for (const bad of [ + "v1", + "radius/1", + "radius", + "latest", + "", + "radius/v1.2.3", + ]) { + assert.throws( + () => parseVersion(bad), + ProtocolError, + `"${bad}" must be rejected`, + ); + } + }); +}); + +describe("compatibility — a host may lag one minor, never a major", () => { + test("the same version, and a host one minor behind, are compatible", () => { + assert.doesNotThrow(() => assertCompatible("radius/v1.2", "radius/v1.2")); + assert.doesNotThrow(() => assertCompatible("radius/v1.2", "radius/v1.1")); + }); + + test("a host AHEAD on the minor is compatible", () => { + // A newer host talking to an older plane is fine for additive changes. + assert.doesNotThrow(() => assertCompatible("radius/v1.1", "radius/v1.2")); + }); + + test("a host two minors behind is NOT compatible", () => { + // "Only one minor" means one. Assuming two is how a compat promise quietly dies. + assert.throws( + () => assertCompatible("radius/v1.3", "radius/v1.1"), + /minors behind/, + ); + }); + + test("a major mismatch is refused in both directions", () => { + assert.throws( + () => assertCompatible("radius/v2.0", "radius/v1.9"), + UnsupportedVersionError, + ); + assert.throws( + () => assertCompatible("radius/v1.9", "radius/v2.0"), + UnsupportedVersionError, + ); + }); + + test("a malformed version is refused, not coerced", () => { + assert.throws( + () => assertCompatible("radius/v1.0", "garbage"), + ProtocolError, + ); + }); +}); + +describe("change classification", () => { + test("a new optional field or record kind is a minor", () => { + assert.equal( + classifyRelease({ version: "radius/v1.1", newOptionalFields: ["scope"] }), + "minor", + ); + assert.equal( + classifyRelease({ version: "radius/v1.1", newRecordKinds: ["handoff"] }), + "minor", + ); + }); + + test("a removal, rename, or semantic change is a MAJOR", () => { + assert.equal( + classifyRelease({ version: "radius/v2.0", removedOrRenamed: ["scope"] }), + "major", + ); + assert.equal( + classifyRelease({ + version: "radius/v2.0", + semanticChanges: ["seq now starts at 1"], + }), + "major", + ); + }); + + test("a TIGHTENED DEFAULT is a MAJOR, with no field changed at all", () => { + // The rule that surprises people, and the reason classifyRelease works on semantics rather + // than shapes. Turning a sandbox on cannot be additive: a host built against v1 defaults must + // fail loudly, not discover it later. + assert.equal( + classifyRelease({ + version: "radius/v2.0", + tightenedDefaults: ["sandbox on by default"], + }), + "major", + ); + }); + + test("an empty release is neither minor nor major", () => { + assert.equal(classifyRelease({ version: "radius/v1.0" }), "none"); + }); +}); + +describe("version bump enforcement", () => { + test("a minor additive change passes", () => { + assert.doesNotThrow(() => + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + newOptionalFields: ["scope"], + }), + ); + }); + + test("a major breaking change passes", () => { + assert.doesNotThrow(() => + assertVersionBump("radius/v1.0", { + version: "radius/v2.0", + tightenedDefaults: ["sandbox on"], + }), + ); + }); + + test("a breaking change shipped as a minor is REFUSED", () => { + // Direction one. The mistake everyone catches, because their own client breaks. + assert.throws( + () => + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + semanticChanges: ["cursor semantics changed"], + }), + BreakingReleaseError, + ); + }); + + test("a tightened default shipped as a minor is REFUSED", () => { + assert.throws( + () => + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + tightenedDefaults: ["sandbox on by default"], + }), + /breaking/, + ); + }); + + test("an additive change shipped as a MAJOR is REFUSED", () => { + // Direction two, and the subtler one. It never breaks anyone; it just forces an upgrade for + // an optional field, until "stable" means nothing. + assert.throws( + () => + assertVersionBump("radius/v1.0", { + version: "radius/v2.0", + newOptionalFields: ["scope"], + }), + /MAJOR version/, + ); + }); + + test("a version that goes backwards is refused", () => { + assert.throws( + () => + assertVersionBump("radius/v2.0", { + version: "radius/v1.9", + newOptionalFields: ["x"], + }), + /backwards/, + ); + }); + + test("the breaking error names what broke", () => { + // "This is breaking" is not actionable; naming the field is. + try { + assertVersionBump("radius/v1.0", { + version: "radius/v1.1", + removedOrRenamed: ["scope"], + semanticChanges: ["seq base changed"], + }); + assert.fail("should have thrown"); + } catch (e) { + assert.ok(e instanceof BreakingReleaseError); + assert.ok( + e.reasons.some((r) => r.includes("scope")), + "names the removed field", + ); + } + }); +}); diff --git a/packages/protocol/tsconfig.json b/packages/protocol/tsconfig.json new file mode 100644 index 0000000..4782c91 --- /dev/null +++ b/packages/protocol/tsconfig.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "outDir": "dist", + "rootDir": "src", + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts"] +} diff --git a/packages/protocol/tsconfig.test.json b/packages/protocol/tsconfig.test.json new file mode 100644 index 0000000..b9039d2 --- /dev/null +++ b/packages/protocol/tsconfig.test.json @@ -0,0 +1,12 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "../config/tsconfig.base.json", + "compilerOptions": { + "noEmit": true, + "declaration": false, + "sourceMap": false, + "lib": ["ES2024"], + "types": ["node"] + }, + "include": ["src/**/*.ts", "test/**/*.ts"] +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 791afb9..89f7263 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -62,8 +62,36 @@ importers: specifier: ^5.9.3 version: 5.9.3 + packages/broker: + dependencies: + '@graycode/protocol': + specifier: workspace:* + version: link:../protocol + devDependencies: + '@types/node': + specifier: ^24 + version: 24.19.0 + tsx: + specifier: ^4.21.0 + version: 4.23.15 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + packages/config: {} + packages/protocol: + devDependencies: + '@types/node': + specifier: ^24 + version: 24.19.0 + tsx: + specifier: ^4.21.0 + version: 4.23.15 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + packages: '@agentclientprotocol/sdk@1.5.0': diff --git a/scripts/check-invariants.mjs b/scripts/check-invariants.mjs index 48d1094..001136e 100644 --- a/scripts/check-invariants.mjs +++ b/scripts/check-invariants.mjs @@ -144,6 +144,23 @@ const ADOPTED_VENDOR_INVARIANTS = [ /** Invariant ids above that are satisfied by the vendored check rather than by prose. */ const VENDORED_INVARIANT_IDS = new Set(["I15"]); +/** + * Invariant ids that are enforced in OUR code, with a test that fails if violated. + * + * Added 2026-09-28 with `packages/protocol`. Until this existed the honest count was "14 of 15 + * are specified only", and a reader could reasonably assume the whole table was prose. + * + * Each entry names the test that would fail if the behaviour regressed, so this set can be + * audited rather than believed: + * I3 expired lease refused on write — packages/protocol/test/lease.test.ts + * I5 stale epoch rejected on write — packages/protocol/test/lease.test.ts + * I14 wall-clock rollback cannot revive a lease — packages/protocol/test/lease.test.ts + * + * A claim is only as good as the test that would catch its removal. If one of these stops being + * enforced, remove it from this set in the same commit that removes the enforcement. + */ +const ENFORCED_INVARIANT_IDS = new Set(["I3", "I5", "I14"]); + const problems = []; const notes = []; @@ -336,13 +353,15 @@ if (existsSync(brokerSrc)) { radiusEnforced = readdirSync(brokerSrc).some((f) => f.endsWith(".ts")); } const specCount = INVARIANTS.filter( - ([id]) => !VENDORED_INVARIANT_IDS.has(id), + ([id]) => !VENDORED_INVARIANT_IDS.has(id) && !ENFORCED_INVARIANT_IDS.has(id), ).length; if (!radiusEnforced) { notes.push( - `${specCount} of ${INVARIANTS.length} invariants (I1–I${specCount}) are SPECIFIED only — ` + - `packages/broker does not exist.\n` + - ` They are not enforced in code. Do not read a pass as a guarantee.\n` + + `${specCount} of ${INVARIANTS.length} invariants are SPECIFIED only — packages/broker does ` + + `not exist.\n` + + ` Those are NOT enforced in code. Do not read a pass as a guarantee.\n` + + ` ${ENFORCED_INVARIANT_IDS.size} (${[...ENFORCED_INVARIANT_IDS].join(", ")}) ARE ` + + `enforced in packages/protocol, with tests that fail if violated.\n` + ` ${ADOPTED_VENDOR_INVARIANTS.length} (${[...VENDORED_INVARIANT_IDS].join(", ")}) ` + `IS checked against real vendored source.`, ); @@ -400,7 +419,8 @@ if (problems.length > 0) { // reader ends up believing 15 things are enforced when 1 is. console.log( `check:invariants OK — ${INVARIANTS.length} invariants mapped ` + - `(${specCount} specified, ${ADOPTED_VENDOR_INVARIANTS.length} verified against vendored source)`, + `(${specCount} specified only, ${ENFORCED_INVARIANT_IDS.size} enforced in code, ` + + `${ADOPTED_VENDOR_INVARIANTS.length} verified against vendored source)`, ); for (const n of notes) console.error(`\n note: ${n}`); console.error(""); diff --git a/scripts/probe-runtimes.mjs b/scripts/probe-runtimes.mjs new file mode 100644 index 0000000..c4be17f --- /dev/null +++ b/scripts/probe-runtimes.mjs @@ -0,0 +1,143 @@ +#!/usr/bin/env node +/** + * Probe the installed agent CLIs and emit a *measured* capability matrix. + * + * `ADAPTERS.md` opens with "verify, don't assume — this table will rot". This script is the + * verification, so the table is a measurement rather than a recollection. It is not in CI: it + * depends on which CLIs happen to be installed, so a missing runtime is reported, not failed. + * + * Why it matters more than a version list. Two claims this project rests on are checkable here: + * + * 1. **"Sessions run YOLO by default."** That is the premise of the entire product. It is + * checkable: if each CLI exposes a permission-bypass flag, and oar passes it, then the + * premise is a fact rather than a quotation from a contract file. + * 2. **The managed-copy invariant (I15) is not specific to k-carrier.** Four of the five + * runtimes ship their own `update`/`upgrade` subcommand. If Radius ever installs and + * manages a copy of one of them — exactly the k-carrier pattern — that self-update is a + * second writer on a component the host believes it owns. `k.managed-copy-never-self-upgrades` + * therefore generalises from our installer to every managed binary, and this probe is how + * we notice when a new one appears. + * + * Run: pnpm probe:runtimes (human-readable) + * pnpm probe:runtimes -- --json (machine-readable, for the matrix check) + */ +import { execFileSync } from "node:child_process"; +import { writeFileSync } from "node:fs"; +import { join } from "node:path"; + +const ROOT = new URL("..", import.meta.url).pathname; +const AS_JSON = process.argv.includes("--json"); + +/** The five runtimes oar ships adapters for (`@botiverse/oar` index.d.ts). */ +const RUNTIMES = ["claude", "codex", "grok", "kimi", "pi"]; + +/** + * Patterns are matched against `--help` output. They are deliberately conservative: a match + * means the CLI *documents* the capability, not that we have verified its enforcement. The + * `documented` prefix in the field names is the honesty — a flag existing is not a sandbox + * working, and `docs/SAFETY.md` §3 is where that distinction is recorded. + */ +const PROBES = { + version: /^\s*(\d+\.\d+\.\d+)/m, + nativeSandbox: + /\b(--sandbox|-s, --sandbox|sandbox mode|workspace-write|danger-full-access)\b/i, + permissionBypass: + /(--dangerously-skip-permissions|--allow-dangerously-skip-permissions|--always-approve|--yolo\b|always-approve)/i, + sandboxNetworkNote: /no internet access/i, + selfUpdate: /^\s+(update|upgrade)\b/m, + workspaceDir: /(--add-dir|workspace directory|working directory)/i, + toolAllowlist: + /(--allowedTools|--disallowedTools|permission (allow|deny) rule|--tools\b)/i, +}; + +function probe(runtime) { + let help = ""; + let installed = true; + let version = null; + try { + help = execFileSync(runtime, ["--help"], { + encoding: "utf8", + timeout: 25_000, + }); + } catch { + installed = false; + } + + if (installed) { + try { + version = execFileSync(runtime, ["--version"], { + encoding: "utf8", + timeout: 25_000, + }) + .toString() + .trim(); + } catch { + version = null; + } + } + + const flags = Object.fromEntries( + Object.entries(PROBES) + .filter(([k]) => k !== "version" && k !== "selfUpdate") + .map(([k, re]) => [k, installed && re.test(help)]), + ); + + return { + runtime, + installed, + version, + // Documented, not enforced. Nothing here proves a sandbox works. + ...flags, + selfUpdates: installed && PROBES.selfUpdate.test(help), + updateCommand: help.match(/^\s+(update|upgrade)\b/m)?.[1] ?? null, + }; +} + +const results = RUNTIMES.map(probe); + +if (AS_JSON) { + writeFileSync( + join(ROOT, "docs/runtimes.probe.json"), + `${JSON.stringify(results, null, 2)}\n`, + ); + console.log(`wrote docs/runtimes.probe.json (${results.length} runtimes)`); + process.exit(0); +} + +const yesNo = (b) => (b === null ? "n/a" : b ? "YES" : "-"); +const pad = (s, n) => String(s ?? "-").padEnd(n); + +console.log( + `\nRuntime capability probe — ${new Date().toISOString().slice(0, 10)}\n`, +); +console.log( + ` ${pad("runtime", 8)}${pad("version", 26)}${pad("sandbox?", 10)}${pad("perm-bypass", 12)}self-update`, +); +console.log(` ${"-".repeat(74)}`); +for (const r of results) { + console.log( + ` ${pad(r.runtime, 8)}${pad(r.version, 26)}${pad(yesNo(r.nativeSandbox), 10)}` + + `${pad(yesNo(r.permissionBypass), 12)}${yesNo(r.selfUpdates)}${ + r.selfUpdates ? ` (${r.updateCommand})` : "" + }`, + ); +} + +const missing = results.filter((r) => !r.installed); +if (missing.length > 0) { + console.log( + `\n not installed: ${missing.map((m) => m.runtime).join(", ")} — reported, not failed`, + ); +} + +const selfUpdating = results.filter((r) => r.selfUpdates); +console.log( + `\n ${selfUpdating.length} of ${results.length} runtimes ship a self-update path. If Radius ` + + `ever manages a copy of one, that is a SECOND WRITER on a component the host believes it ` + + `owns — the same shape as k.managed-copy-never-self-upgrades (I15), which therefore ` + + `generalises beyond k-carrier.`, +); +console.log( + `\n A documented flag is not a working sandbox. This measures what each CLI CLAIMS to support;\n` + + ` the isolation it actually enforces is still unmeasured — see SAFETY.md §3.\n`, +);