From 414d3e1fcfaffb5228c177961435cd55ec1b9996 Mon Sep 17 00:00:00 2001 From: Mother Seara Date: Wed, 9 Sep 2026 21:54:18 +0900 Subject: [PATCH] fix: require verified seals and explicit bound resolutions at gates --- CHANGELOG.md | 12 +++ README.md | 28 +++++- docs/STACK_CANONICAL.md | 6 ++ mirror_stack_mcp/__init__.py | 2 +- mirror_stack_mcp/gate.py | 105 +++++++++++++--------- mirror_stack_mcp/integrity.py | 49 +++++++++++ mirror_stack_mcp/server.py | 31 ++++--- mirror_stack_mcp/verify.py | 34 ++++---- pyproject.toml | 2 +- tests/test_gate.py | 18 +++- tests/test_integrity_boundaries.py | 134 +++++++++++++++++++++++++++++ tests/test_server.py | 18 +++- tests/test_verify.py | 4 +- 13 files changed, 361 insertions(+), 82 deletions(-) create mode 100644 mirror_stack_mcp/integrity.py create mode 100644 tests/test_integrity_boundaries.py diff --git a/CHANGELOG.md b/CHANGELOG.md index aceb71a..65e95fd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,17 @@ # Changelog +## [0.2.14] β€” 2026-09-09 + +- Gate and outsider CLI recompute MIRROR-SPEC content hashes and links from a + single snapshot. Missing, empty, malformed, duplicate-key and tampered ledgers fail closed. +- Publish requires a reasoned retraction or explicit `action=result` with + `payload.status`, `payload.summary`, and `payload.prereg_seal` binding the first registration. + Merely starting work no longer resolves a claim. Negative results remain publishable. +- Expose verification depth and unverified content truth, author identity, + external time and independent reproduction. A separately checked OTS proof is not + presented as proof of this ledger's precedence without checking its binding. +- Migration: append a real explicit result; do not rewrite historical sealed actions. + All notable changes to mirror-stack-mcp are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.0.0/). diff --git a/README.md b/README.md index bf39ed8..92b735c 100644 --- a/README.md +++ b/README.md @@ -36,6 +36,30 @@ Windsurf, any MCP client. ## Tools (19) +### Verification depth and gate migration (0.2.14) + +`mirror-stack-verify` now recomputes both content hashes and chain links; pointer-only +fixtures do not certify integrity. Legacy 16-hex seals are accepted with a warning. +`stack_verify_all` labels checks `HASH_RECOMPUTED`, `LOCAL_SNAPSHOT`, or `HEAD_WITNESS`. +None certifies content truth, author identity, independent reproduction or external time. +The outsider CLI checks an optional OTS proof separately; ledger-to-proof binding is +not checked, so it does not claim this ledger's external-clock precedence. + +The compute/publish gate verifies ledger integrity before reading the decision facts. +Publish requires a reasoned retraction, or an explicit result bound to the original seal: + +```python +am_record(ledger_path="actions.jsonl", agent="researcher", action="result", target="claim-id", + payload={"status": "fail", "summary": "Measured value did not meet the preregistered bar", + "prereg_seal": ""}) +``` + +Allowed statuses are `pass`, `fail`, and `inconclusive`. A started action does not +resolve a claim. Append a new result instead of rewriting historical records. GO permits +reporting a resolved result, including a failure; it does not certify scientific success. + +### Tool reference + | Tool | Mirror | Does | |---|---|---| | `mm_preregister` | πŸͺž claims | seal a claim + kill-condition **before** measuring (response carries an auto seal-quality lint) | @@ -160,7 +184,7 @@ into the action site: kill-condition* is sealed. Wire it into your training/experiment launcher so it refuses to spend compute on an unsealed claim. - `mm_preflight(ledger, claim_id, gate="publish", am_ledger=…)` β†’ **BLOCK** unless the *result* - is also sealed (a retraction, or `am_record(target=claim_id)`). Wire it into a pre-commit / + is also sealed (a reasoned retraction, or the explicit bound result above). Wire it into a pre-commit / pre-publish hook so unresolved claims can't ship. The MCP only *judges* GO/BLOCK β€” **your** launcher/hook does the actual blocking. The server @@ -174,7 +198,7 @@ the part that actually exits non-zero, so a shell can do the blocking the MCP ca mirror-stack-gate compute --ledger L.jsonl --claim my_claim && python run.py # exits 1 (run.py never starts) unless a kill-conditioned preregistration is sealed mirror-stack-gate publish --ledger L.jsonl --claim my_claim --am-ledger A.jsonl -# exits 1 unless the result is sealed too (a retraction or am_record(target=claim)) +# exits 1 unless a reasoned retraction or an explicit bound result is sealed too ``` For git, drop in [`hooks/pre-commit.sample`](hooks/pre-commit.sample): it runs the publish gate diff --git a/docs/STACK_CANONICAL.md b/docs/STACK_CANONICAL.md index 8db78c0..cb8589f 100644 --- a/docs/STACK_CANONICAL.md +++ b/docs/STACK_CANONICAL.md @@ -1,5 +1,11 @@ # Mirror Stack β€” canonical surfaces & the one shared primitive +Since 0.2.14, `check_chain()` remains the linkage-only compatibility API, but the +outsider CLI and gate use `integrity.read_verified()` for a single snapshot with +both links and MIRROR-SPEC content hashes verified. Signature identity, external +time and content truth remain outside this check. The directory orchestrator +explicitly labels its automatically included ledgers as linkage-only. + Two repos make up the running stack. They are **different surfaces, not duplicates** β€” keep them separate. This is the canonical map of who owns what, so "which one is authoritative?" has a written answer. diff --git a/mirror_stack_mcp/__init__.py b/mirror_stack_mcp/__init__.py index 178edbb..2aa8599 100644 --- a/mirror_stack_mcp/__init__.py +++ b/mirror_stack_mcp/__init__.py @@ -1,2 +1,2 @@ """πŸͺžπŸ”ŽπŸͺͺ Mirror Stack unified MCP server.""" -__version__ = "0.2.13" +__version__ = "0.2.14" diff --git a/mirror_stack_mcp/gate.py b/mirror_stack_mcp/gate.py index 1801e42..5e961d9 100644 --- a/mirror_stack_mcp/gate.py +++ b/mirror_stack_mcp/gate.py @@ -12,17 +12,16 @@ publish (resolution-before-publish), e.g. a git pre-commit hook: mirror-stack-gate publish --ledger L.jsonl --claim my_claim --am-ledger A.jsonl - # exits 1 unless a retraction or an am_record(target=my_claim) is sealed + # requires a reasoned retraction or action=result bound by payload.prereg_seal The same `decide()` backs server.mm_preflight, so the agent-facing tool and the shell enforcer can never drift apart. """ -import json -import os import sys +from .integrity import read_verified -def scan_claim(ledger_path, claim_id): +def scan_claim(ledger_path, claim_id, entries=None): """Return (prereg_entry_or_None, retracted_bool, leaked_entry_or_None) for claim_id. leaked_entry is a preregistration that carries NO kill fields but whose `metric` @@ -31,38 +30,46 @@ def scan_claim(ledger_path, claim_id): the gate report the real reason instead of a misleading 'no preregistration'. """ prereg, retracted, leaked = None, False, None - if os.path.exists(ledger_path): - with open(ledger_path, encoding="utf-8") as fh: - for line in fh: - line = line.strip() - if not line: - continue - try: - e = json.loads(line) - except json.JSONDecodeError: - continue - if e.get("claim_id") != claim_id: - continue - if e.get("_type") == "retraction": - retracted = True - elif e.get("_type") is None and ("kill_threshold" in e or "kill_condition" in e) \ - and e.get("metric") != "protocol_amendment": - prereg = e - elif e.get("_type") is None and e.get("metric") not in (None, "protocol_amendment"): - from measure_mirror import mm - if leaked is None and mm._looks_like_kill_prose(e.get("metric", "")): - leaked = e + seen_registration = False + if entries is None: + entries, error = read_verified(ledger_path) + if error: + return None, False, None + for e in entries: + if e.get("claim_id") != claim_id: + continue + if e.get("_type") == "retraction": + retracted = bool(prereg and isinstance(e.get("reason"), str) + and e["reason"].strip()) or retracted + elif e.get("_type") is None and e.get("metric") != "protocol_amendment": + if seen_registration: + continue + seen_registration = True + if "kill_threshold" in e or "kill_condition" in e: + prereg = e # First-write wins, even when the first registration is invalid. + else: + from measure_mirror import mm + if isinstance(e.get("metric"), str) and mm._looks_like_kill_prose(e["metric"]): + leaked = e return prereg, retracted, leaked def decide(ledger_path, claim_id, gate="compute", am_ledger=None, reported_acc=None): """Pure GO/BLOCK decision. Returns {decision, gate, claim_id, reasons, checks}.""" - prereg, retracted, leaked = scan_claim(ledger_path, claim_id) + entries, error = read_verified(ledger_path) + prereg, retracted, leaked = scan_claim(ledger_path, claim_id, entries) checks: list[str] = [] def out(decision, reasons): return {"decision": decision, "gate": gate, "claim_id": claim_id, - "reasons": reasons, "checks": checks} + "reasons": reasons, "checks": checks, + "verification": {"depth": "HASH_RECOMPUTED" if not error else "UNVERIFIED", + "content_truth": "unverified", "external_time": "unverified", + "independent_reproduction": "unverified"}} + + if error: + return out("BLOCK", ["claims ledger integrity failed: " + error]) + checks.append("claims ledger: linkage and all content hashes verified") if prereg is None: if leaked is not None: @@ -77,6 +84,8 @@ def out(decision, reasons): checks.append("preregistration: sealed" + ("" if has_kill else " (NO kill-condition)")) if gate == "compute": + if retracted: + return out("BLOCK", ["claim is retracted; use a new preregistration"]) if not has_kill: return out("BLOCK", ["preregistration has no kill-condition (unfalsifiable) β€” " "add one before spending compute"]) @@ -84,7 +93,10 @@ def out(decision, reasons): # automated checks can't do their job β€” compute would be spent against a # meaningless bar. WARN/INFO inform but don't block. from measure_mirror import mm - lint = mm._preseal_lint(prereg) + try: + lint = mm._preseal_lint(prereg) + except (TypeError, ValueError, AttributeError, KeyError) as exc: + return out("BLOCK", [f"preregistration cannot be evaluated: {exc}"]) fails = [f for f in lint if f.level == "FAIL"] warns = [f for f in lint if f.level == "WARN"] if warns: @@ -98,25 +110,34 @@ def out(decision, reasons): resolved = retracted if retracted: checks.append("resolution: retraction sealed") - if not resolved and am_ledger and os.path.exists(am_ledger): - with open(am_ledger, encoding="utf-8") as fh: - for line in fh: - try: - a = json.loads(line) - except json.JSONDecodeError: - continue - if a.get("_type") == "action" and a.get("target") == claim_id: - resolved = True - checks.append("resolution: am_record(target) sealed") - break + if not resolved and am_ledger: + actions, action_error = read_verified(am_ledger) + if action_error: + return out("BLOCK", ["action ledger integrity failed: " + action_error]) + checks.append("action ledger: linkage and all content hashes verified") + for a in actions: + payload = a.get("payload") + if (a.get("_type") == "action" and a.get("target") == claim_id + and a.get("action") == "result" and isinstance(payload, dict) + and payload.get("status") in ("pass", "fail", "inconclusive") + and isinstance(payload.get("summary"), str) and payload["summary"].strip() + and payload.get("prereg_seal") == prereg["seal"]): + resolved = True + checks.append("resolution: explicit result bound to verified preregistration") + break if not resolved: return out("BLOCK", ["no sealed resolution β€” seal the result " - "(am_record target=claim_id, or mm_retract) before publishing. " + "(am_record action=result, target=claim_id, payload with status, " + "summary, prereg_seal; or mm_retract with reason) before publishing. " "Prose doesn't count."]) if reported_acc is not None: from measure_mirror import mm - checks.append(str(mm.falsifiability_check(ledger_path, claim_id, reported_acc=reported_acc))) - return out("GO", ["sealed preregistration + sealed resolution"]) + # Evaluate the already-verified snapshot, not a second mutable file read. + finding = mm._falsifiability_eval(prereg, reported_acc) + checks.append(str(finding)) + checks.append("publication may report a negative result; GO is NOT claim success") + return out("GO", ["verified preregistration + explicit sealed resolution; " + "publication permitted, content truth not certified"]) return out("BLOCK", [f"unknown gate '{gate}' β€” use 'compute' or 'publish'"]) diff --git a/mirror_stack_mcp/integrity.py b/mirror_stack_mcp/integrity.py new file mode 100644 index 0000000..1bf1853 --- /dev/null +++ b/mirror_stack_mcp/integrity.py @@ -0,0 +1,49 @@ +"""Read and verify one immutable-in-memory MIRROR-SPEC ledger snapshot. + +Hash integrity is not signature identity, external timing, or content truth. +Both the gate and outsider CLI use this path; neither certifies pointer linkage alone. +""" +import hashlib +import json +from pathlib import Path + + +def _object(pairs): + out = {} + for key, value in pairs: + if key in out: + raise ValueError(f"duplicate JSON key: {key}") + out[key] = value + return out + + +def read_verified(path): + """Return (entries, error). Fail closed on missing, empty or corrupt input. + + Reads once so decisions inspect exactly the snapshot whose hashes were checked. + Accepts legacy 16-hex seals at their original, weaker assurance level. + """ + try: + raw = Path(path).read_text(encoding="utf-8") + entries = [json.loads(line, object_pairs_hook=_object) + for line in raw.splitlines() if line.strip()] + if not entries: + return [], "ledger is empty; nothing verified" + previous = "genesis" + for i, entry in enumerate(entries, 1): + if not isinstance(entry, dict): + return [], f"entry {i}: JSON object required" + link, seal = entry.get("prev_seal"), entry.get("seal") + if not isinstance(link, str) or (link.lower() != "genesis" if i == 1 else link != previous): + return [], f"entry {i}: chain linkage broken" + if not isinstance(seal, str) or len(seal) not in (16, 64): + return [], f"entry {i}: missing or invalid seal" + body = {k: v for k, v in entry.items() if k not in ("seal", "sig")} + digest = hashlib.sha256(json.dumps(body, sort_keys=True, ensure_ascii=False, + allow_nan=False).encode("utf-8")).hexdigest() + if seal != digest[:len(seal)]: + return [], f"entry {i}: seal mismatch; content modified" + previous = seal + return entries, None + except (OSError, UnicodeError, ValueError, TypeError, RecursionError) as exc: + return [], f"ledger cannot be verified: {exc}" diff --git a/mirror_stack_mcp/server.py b/mirror_stack_mcp/server.py index 7578d46..fd061fb 100644 --- a/mirror_stack_mcp/server.py +++ b/mirror_stack_mcp/server.py @@ -137,7 +137,8 @@ def _compact(findings): REMINDERS = { "mm_preregister": "πŸͺž Sealed. Not done until the RESULT is sealed too β€” " - "am_record(target=claim_id) on a verdict, or mm_retract if falsified. Prose doesn't " + "am_record(action=result, target=claim_id, payload={status, summary, prereg_seal}) " + "on a verdict, or mm_retract if falsified. Prose doesn't " "count. Your kill_condition is the stop-loss; if big compute follows, seal first, then run.", "mm_verify": _VERIFY, "mm_audit": _VERIFY, @@ -373,7 +374,11 @@ def mm_preflight(ledger_path: str, claim_id: str, gate: str = "compute", gate="compute": GO only if a sealed preregistration WITH a kill-condition exists for claim_id (enforces seal-before-compute). gate="publish": additionally GO only if a RESOLUTION is sealed β€” a retraction in - ledger_path, or an am_record(target=claim_id) in am_ledger. + ledger_path (with reason), or am_record(action=result, target=claim_id) + with payload status=pass/fail/inconclusive, nonempty summary, + and prereg_seal matching the first verified registration. + Both ledgers must pass hash and linkage verification. GO authorizes publication + of a resolved result, including failures; it does NOT certify claim success. This is a PRIMITIVE: the MCP returns GO/BLOCK; YOUR script must do the actual blocking (the MCP cannot intercept external compute or commits β€” that is by design). The shell enforcer that DOES exit non-zero is `mirror-stack-gate` (mirror_stack_mcp.gate); both @@ -442,15 +447,15 @@ def stack_verify_all(mm_ledger: str, anchor_dir: str | None = None, def add(level, layer, name, msg): nonlocal ok ok = ok and level - out.append({"ok": level, "layer": layer, "name": name, "msg": msg}) + depth = {"L1 chain": "HASH_RECOMPUTED", "L3 anchor": "LOCAL_SNAPSHOT", + "L2 witness": "HEAD_WITNESS"}[layer] + out.append({"ok": level, "layer": layer, "name": name, "msg": msg, + "depth": depth}) - findings = mm.verify_chain(mm_ledger) - bad = [str(f) for f in findings if getattr(f, "level", "OK") not in ("OK", "INFO")] - # Say how many seals were checked. "seals valid" is also true of an empty ledger. - n_entries = sum(1 for l in Path(mm_ledger).read_text(encoding="utf-8", - errors="replace").splitlines() if l.strip()) - add(not bad, "L1 chain", Path(mm_ledger).name, - f"seals valid ({n_entries} entries checked)" if not bad else str(bad)) + from .integrity import read_verified + entries, error = read_verified(mm_ledger) + add(error is None, "L1 chain", Path(mm_ledger).name, + error or f"seals valid ({len(entries)} entries checked)") if anchor_dir: for af in sorted(Path(anchor_dir).glob("anchor_*.json")): @@ -486,7 +491,11 @@ def add(level, layer, name, msg): "scope": {"mm_ledger": Path(mm_ledger).name, "layers_run": sorted({c["layer"] for c in out}), "layers_not_requested": [] if anchor_dir else ["L3 anchor"], - "layers_skipped": skipped}} + "layers_skipped": skipped, + "external_time": "unverified", + "author_identity": "unverified", + "independent_reproduction": "unverified", + "content_truth": "unverified"}} if skipped: # `skipped` means REQUESTED-but-did-not-run, which is the defect this fixes. # Not passing `anchor_dir` at all is not that β€” you did not ask for L3, so it is diff --git a/mirror_stack_mcp/verify.py b/mirror_stack_mcp/verify.py index 959f5a3..fce95f1 100644 --- a/mirror_stack_mcp/verify.py +++ b/mirror_stack_mcp/verify.py @@ -4,22 +4,22 @@ mirror-stack-verify LEDGER.jsonl [--ots PROOF.ots] [--explorer URL] -Two independent confirmations, both reproducible by anyone: +Verification scope: - 1. chain β€” recompute the prev_sealβ†’seal linkage of the ledger. A break means an - entry was inserted, deleted, or reordered after sealing. (no network, - stdlib only β€” works on any mirror ledger.) + 1. integrity β€” recompute both links AND content hashes from one ledger snapshot. + No network; stdlib-only MIRROR-SPEC verification. Optional identity + signatures are NOT verified; legacy 16-hex seals remain weaker. 2. bitcoin β€” if an OpenTimestamps proof is given, cross-check its Bitcoin block - against a PUBLIC block explorer: the ledger head existed before that - block's time. This is the "don't trust us" part β€” the clock is - Bitcoin's and the lookup is a third party's, not ours. (needs the - `ots` CLI + network.) + against a PUBLIC block explorer. This CLI does NOT check binding + between that proof and this ledger, so ledger precedence remains + UNVERIFIED. (needs the `ots` CLI + network.) -Honest scope: this proves INTEGRITY (not tampered) and PRECEDENCE (not backdated). +Honest scope: this checks hash INTEGRITY, not external-clock PRECEDENCE. It does NOT prove the content is true, nor that an independent judge witnessed it. """ import argparse import sys +from .integrity import read_verified OK, FAIL, WARN = "βœ…", "❌", "⚠️" @@ -57,8 +57,12 @@ def main(argv=None): a = ap.parse_args(argv) print("=== πŸͺžπŸ”ŽπŸͺͺ mirror-stack-verify (you recompute β€” you don't trust us) ===") - ok_chain, msg = check_chain(a.ledger) - print(f"{OK if ok_chain else FAIL} [chain] {msg}") + entries, error = read_verified(a.ledger) + ok_chain = error is None + msg = error or f"{len(entries)} entries: linkage and SHA-256 seals recomputed" + print(f"{OK if ok_chain else FAIL} [HASH_RECOMPUTED] {msg}") + if entries and any(len(e["seal"]) == 16 for e in entries): + print(f"{WARN} legacy 16-hex seals: weaker collision resistance") ok_btc = True if a.ots: @@ -76,10 +80,10 @@ def main(argv=None): print(f"{WARN} [bitcoin] skipped β€” pass --ots PROOF.ots for the external-clock check") good = ok_chain and ok_btc - print(f"=== verdict: {'CONFIRMED' if good else 'NOT CONFIRMED'} " - f"(integrity{' + precedence' if a.ots and ok_btc else ''}) ===") - print("scope: proves not-tampered" + (" + not-backdated" if a.ots else "") + - "; NOT content truth, NOT an independent judging witness.") + print(f"=== verdict: {'CONFIRMED' if good else 'NOT CONFIRMED'} (hash integrity) ===") + print("scope: hash integrity only; NOT author identity, content truth or independent reproduction.") + print("external time: ledger precedence UNVERIFIED β€” " + "a supplied Bitcoin proof is checked separately; ledger-to-proof binding is not checked here.") return 0 if good else 1 diff --git a/pyproject.toml b/pyproject.toml index f2a6bde..81644bb 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "mirror-stack-mcp" -version = "0.2.13" +version = "0.2.14" description = "Unified MCP server for the Mirror Stack β€” claims, actions, provenance + verify-all in one server" readme = "README.md" requires-python = ">=3.10" diff --git a/tests/test_gate.py b/tests/test_gate.py index 4cc1bf0..5800d65 100644 --- a/tests/test_gate.py +++ b/tests/test_gate.py @@ -5,6 +5,7 @@ the same decide(), so the agent tool and the shell enforcer can't drift. """ import json +import hashlib from mirror_stack_mcp import gate @@ -12,7 +13,14 @@ def _w(p, entries): - p.write_text("\n".join(json.dumps(e) for e in entries) + "\n", encoding="utf-8") + sealed, prev = [], "genesis" + for entry in entries: + e = {**entry, "prev_seal": prev} + e["seal"] = hashlib.sha256(json.dumps(e, sort_keys=True, ensure_ascii=False).encode()).hexdigest() + sealed.append(e) + prev = e["seal"] + p.write_text("\n".join(json.dumps(e) for e in sealed) + "\n", encoding="utf-8") + return sealed # ── decide(): compute gate ──────────────────────────────────────────────────── @@ -43,14 +51,16 @@ def test_publish_blocks_unresolved(tmp_path): def test_publish_go_with_retraction(tmp_path): led = tmp_path / "l.jsonl" - _w(led, [PRE_KILL, {"_type": "retraction", "claim_id": "c1"}]) + _w(led, [PRE_KILL, {"_type": "retraction", "claim_id": "c1", "reason": "negative result"}]) assert gate.decide(str(led), "c1", "publish")["decision"] == "GO" def test_publish_go_with_am_record(tmp_path): led, am = tmp_path / "l.jsonl", tmp_path / "a.jsonl" - _w(led, [PRE_KILL]) - _w(am, [{"_type": "action", "target": "c1"}]) + pre = _w(led, [PRE_KILL])[0] + _w(am, [{"_type": "action", "target": "c1", "action": "result", + "payload": {"status": "fail", "summary": "threshold not met", + "prereg_seal": pre["seal"]}}]) assert gate.decide(str(led), "c1", "publish", am_ledger=str(am))["decision"] == "GO" diff --git a/tests/test_integrity_boundaries.py b/tests/test_integrity_boundaries.py new file mode 100644 index 0000000..41f4550 --- /dev/null +++ b/tests/test_integrity_boundaries.py @@ -0,0 +1,134 @@ +import hashlib +import json + +import pytest + +from mirror_stack_mcp import gate, verify +from mirror_stack_mcp.integrity import read_verified + + +def write_chain(path, entries, legacy=False): + sealed, previous = [], "genesis" + for entry in entries: + row = {**entry, "prev_seal": previous} + row["seal"] = hashlib.sha256(json.dumps(row, sort_keys=True, ensure_ascii=False).encode()).hexdigest() + if legacy: + row["seal"] = row["seal"][:16] + previous = row["seal"] + sealed.append(row) + path.write_text("\n".join(json.dumps(e) for e in sealed) + "\n") + return sealed + + +PRE = {"claim_id": "c", "metric": "acc", "kill_condition": "kill if below baseline"} + + +@pytest.mark.parametrize("text", ["", "not json", "[]", "null", "42", + json.dumps(PRE), '{"seal":"a","seal":"b"}']) +def test_malformed_or_unsealed_ledger_blocks(tmp_path, text): + path = tmp_path / "claims.jsonl" + path.write_text(text) + assert gate.decide(str(path), "c")["decision"] == "BLOCK" + assert read_verified(path)[1] + + +def test_missing_ledger_blocks(tmp_path): + assert gate.decide(str(tmp_path / "missing"), "c")["decision"] == "BLOCK" + + +@pytest.mark.parametrize("legacy", [False, True]) +def test_real_seal_passes_but_content_edit_fails(tmp_path, legacy, capsys): + path = tmp_path / "claims.jsonl" + rows = write_chain(path, [PRE], legacy) + assert verify.main([str(path)]) == 0 + assert "ledger precedence UNVERIFIED" in capsys.readouterr().out + rows[0]["metric"] = "changed-without-resealing" + path.write_text(json.dumps(rows[0]) + "\n") + assert verify.check_chain(str(path))[0] # linkage alone still passes + assert verify.main([str(path)]) == 1 + assert gate.decide(str(path), "c")["decision"] == "BLOCK" + + +@pytest.mark.parametrize("status", ["pass", "fail", "inconclusive"]) +def test_explicit_bound_results_can_publish(tmp_path, status): + claims, actions = tmp_path / "claims", tmp_path / "actions" + pre = write_chain(claims, [PRE])[0] + write_chain(actions, [{"_type": "action", "action": "result", "target": "c", + "payload": {"status": status, "summary": "observed outcome", + "prereg_seal": pre["seal"]}}]) + result = gate.decide(str(claims), "c", "publish", str(actions)) + assert result["decision"] == "GO" + assert result["verification"]["content_truth"] == "unverified" + + +@pytest.mark.parametrize("change", ["started", "wrong-seal", "empty-summary", "unknown-status", "tampered"]) +def test_invalid_resolution_blocks(tmp_path, change): + claims, actions = tmp_path / "claims", tmp_path / "actions" + pre = write_chain(claims, [PRE])[0] + row = {"_type": "action", "action": "result", "target": "c", + "payload": {"status": "pass", "summary": "observed outcome", "prereg_seal": pre["seal"]}} + if change == "started": + row["action"] = "started" + elif change == "wrong-seal": + row["payload"]["prereg_seal"] = "another registration" + elif change == "empty-summary": + row["payload"]["summary"] = " " + elif change == "unknown-status": + row["payload"]["status"] = "started" + sealed = write_chain(actions, [row]) + if change == "tampered": + sealed[0]["payload"]["summary"] = "changed" + actions.write_text(json.dumps(sealed[0]) + "\n") + assert gate.decide(str(claims), "c", "publish", str(actions))["decision"] == "BLOCK" + + +def test_retraction_requires_reason_and_blocks_compute(tmp_path): + claims = tmp_path / "claims" + write_chain(claims, [PRE, {"_type": "retraction", "claim_id": "c"}]) + assert gate.decide(str(claims), "c", "publish")["decision"] == "BLOCK" + write_chain(claims, [PRE, {"_type": "retraction", "claim_id": "c", "reason": "negative result"}]) + assert gate.decide(str(claims), "c", "publish")["decision"] == "GO" + assert gate.decide(str(claims), "c", "compute")["decision"] == "BLOCK" + + +def test_first_registration_wins(tmp_path): + claims = tmp_path / "claims" + write_chain(claims, [{**PRE, "kill_condition": ""}, PRE]) + assert gate.decide(str(claims), "c")["decision"] == "BLOCK" + write_chain(claims, [{"claim_id": "c", "metric": "acc"}, PRE]) + assert gate.decide(str(claims), "c")["decision"] == "BLOCK" + + +def test_unsigned_action_cannot_resolve(tmp_path): + claims, actions = tmp_path / "claims", tmp_path / "actions" + pre = write_chain(claims, [PRE])[0] + actions.write_text(json.dumps({"_type": "action", "action": "result", "target": "c", + "payload": {"status": "pass", "summary": "done", + "prereg_seal": pre["seal"]}})) + assert gate.decide(str(claims), "c", "publish", str(actions))["decision"] == "BLOCK" + + +def test_decision_reads_claims_snapshot_only_once(tmp_path, monkeypatch): + from pathlib import Path + path = tmp_path / "claims" + write_chain(path, [PRE]) + read_text = Path.read_text + calls = [] + + def counted(self, *args, **kwargs): + if self == path: + calls.append(self) + return read_text(self, *args, **kwargs) + + monkeypatch.setattr(Path, "read_text", counted) + assert gate.decide(str(path), "c")["decision"] == "GO" + assert calls == [path] + + +def test_mcp_empty_ledger_is_not_success(tmp_path): + from mirror_stack_mcp.server import stack_verify_all + claims = tmp_path / "claims" + claims.write_text("") + result = stack_verify_all(str(claims)) + assert result["ok"] is False + assert result["scope"]["independent_reproduction"] == "unverified" diff --git a/tests/test_server.py b/tests/test_server.py index fcd698f..46503b0 100644 --- a/tests/test_server.py +++ b/tests/test_server.py @@ -37,7 +37,15 @@ def _registered(): def _write(path, entries): - path.write_text("\n".join(json.dumps(e) for e in entries) + "\n", encoding="utf-8") + import hashlib + sealed, prev = [], "genesis" + for entry in entries: + e = {**entry, "prev_seal": prev} + e["seal"] = hashlib.sha256(json.dumps(e, sort_keys=True, ensure_ascii=False).encode()).hexdigest() + sealed.append(e) + prev = e["seal"] + path.write_text("\n".join(json.dumps(e) for e in sealed) + "\n", encoding="utf-8") + return sealed # ── registration / defaults ─────────────────────────────────────────────────── @@ -141,15 +149,17 @@ def test_preflight_publish_go_with_retraction(tmp_path): led = tmp_path / "mm.jsonl" _write(led, [ {"claim_id": "c1", "metric": "acc", "kill_threshold": {"below": 0.5}}, - {"_type": "retraction", "claim_id": "c1"}, + {"_type": "retraction", "claim_id": "c1", "reason": "negative result"}, ]) assert s.mm_preflight(str(led), "c1", gate="publish")["decision"] == "GO" def test_preflight_publish_go_with_am_record(tmp_path): led, am = tmp_path / "mm.jsonl", tmp_path / "am.jsonl" - _write(led, [{"claim_id": "c1", "metric": "acc", "kill_threshold": {"below": 0.5}}]) - _write(am, [{"_type": "action", "target": "c1"}]) + pre = _write(led, [{"claim_id": "c1", "metric": "acc", "kill_threshold": {"below": 0.5}}])[0] + _write(am, [{"_type": "action", "target": "c1", "action": "result", + "payload": {"status": "fail", "summary": "threshold not met", + "prereg_seal": pre["seal"]}}]) r = s.mm_preflight(str(led), "c1", gate="publish", am_ledger=str(am)) assert r["decision"] == "GO" diff --git a/tests/test_verify.py b/tests/test_verify.py index ee8dc2f..34988ca 100644 --- a/tests/test_verify.py +++ b/tests/test_verify.py @@ -46,10 +46,10 @@ def test_chain_empty(tmp_path): assert not ok -def test_main_intact_no_ots_confirms(tmp_path): +def test_main_linkage_only_does_not_confirm_integrity(tmp_path): p = tmp_path / "l.jsonl" _w(p, CHAIN) - assert verify.main([str(p)]) == 0 # chain ok, bitcoin skipped β†’ integrity confirmed + assert verify.main([str(p)]) == 1 # pointers alone are not hash integrity def test_main_broken_chain_fails(tmp_path):