diff --git a/.2119/verdicts/REQ-003.1.10--24bb6486f6c2.json b/.2119/verdicts/REQ-003.1.10--24bb6486f6c2.json deleted file mode 100644 index 37391a9..0000000 --- a/.2119/verdicts/REQ-003.1.10--24bb6486f6c2.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "reviewId": "REQ-003.1.10--24bb6486f6c2", - "requirementId": "REQ-003.1.10", - "hash": "24bb6486f6c2", - "verdict": "pass", - "summary": "rigor test runs review and asserts the generated instruction file contains 'Counterexample obligation', matches 'nearest violating input', 'not a pass', and the bad-requirement clause; string matching is the correct verification for a text-generation requirement and would fail if the generator dropped the directive.", - "timestamp": "2026-07-10T22:59:01.487Z" -} diff --git a/.2119/verdicts/REQ-003.1.10.json b/.2119/verdicts/REQ-003.1.10.json new file mode 100644 index 0000000..700226c --- /dev/null +++ b/.2119/verdicts/REQ-003.1.10.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.1.10--d5d332c5e39d", + "requirementId": "REQ-003.1.10", + "hash": "d5d332c5e39d", + "verdict": "pass", + "summary": "Production review packets require a concrete violating change and detection check; removing either directive fails, while whitespace reformatting stays green.", + "timestamp": "2026-09-08T19:28:42.803Z" +} diff --git a/.2119/verdicts/REQ-003.1.11--c6f7a05da4f8.json b/.2119/verdicts/REQ-003.1.11--c6f7a05da4f8.json deleted file mode 100644 index 828601a..0000000 --- a/.2119/verdicts/REQ-003.1.11--c6f7a05da4f8.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "reviewId": "REQ-003.1.11--c6f7a05da4f8", - "requirementId": "REQ-003.1.11", - "hash": "c6f7a05da4f8", - "verdict": "pass", - "summary": "Instruction generator asserts the generated review file carries the 'bad requirement honestly tested is still a bad requirement' clause (the Judge-the-requirement-too directive); a mutant dropping that block fails the test.", - "timestamp": "2026-07-10T22:59:20.283Z" -} diff --git a/.2119/verdicts/REQ-003.1.11.json b/.2119/verdicts/REQ-003.1.11.json new file mode 100644 index 0000000..5761250 --- /dev/null +++ b/.2119/verdicts/REQ-003.1.11.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.1.11--53fcf20cd4f4", + "requirementId": "REQ-003.1.11", + "hash": "53fcf20cd4f4", + "verdict": "pass", + "summary": "Production review packets require findings for ambiguous, untestable, or mechanism-only requirements; dropping that directive fails while reflowing it stays green.", + "timestamp": "2026-09-08T19:28:42.978Z" +} diff --git a/.2119/verdicts/REQ-003.1.12.json b/.2119/verdicts/REQ-003.1.12.json new file mode 100644 index 0000000..e268c31 --- /dev/null +++ b/.2119/verdicts/REQ-003.1.12.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.1.12--f700d391a9a3", + "requirementId": "REQ-003.1.12", + "hash": "f700d391a9a3", + "verdict": "pass", + "summary": "Production review packets require a legitimate meaning-preserving change and green-test confirmation; removing either clause fails while line reflow stays green.", + "timestamp": "2026-09-08T19:28:43.178Z" +} diff --git a/.2119/verdicts/REQ-003.1.13.json b/.2119/verdicts/REQ-003.1.13.json new file mode 100644 index 0000000..16b3185 --- /dev/null +++ b/.2119/verdicts/REQ-003.1.13.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.1.13--be56cbb59e45", + "requirementId": "REQ-003.1.13", + "hash": "be56cbb59e45", + "verdict": "pass", + "summary": "Production review packets allow shared evidence and request cross-reference rather than duplicate tests; dropping that policy fails while whitespace changes stay green.", + "timestamp": "2026-09-08T19:28:43.372Z" +} diff --git a/.2119/verdicts/REQ-003.1.14.json b/.2119/verdicts/REQ-003.1.14.json new file mode 100644 index 0000000..58f89b3 --- /dev/null +++ b/.2119/verdicts/REQ-003.1.14.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.1.14--a21d543aea69", + "requirementId": "REQ-003.1.14", + "hash": "a21d543aea69", + "verdict": "pass", + "summary": "Production review packets reject irrelevant pins while preserving delivered-text, nonvacuous inventory, and maintainable snapshot evidence; reflow remains green.", + "timestamp": "2026-09-08T19:28:43.543Z" +} diff --git a/.2119/verdicts/REQ-003.1.15.json b/.2119/verdicts/REQ-003.1.15.json new file mode 100644 index 0000000..d621a5f --- /dev/null +++ b/.2119/verdicts/REQ-003.1.15.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.1.15--22863c9c36d2", + "requirementId": "REQ-003.1.15", + "hash": "22863c9c36d2", + "verdict": "pass", + "summary": "Production review packets distinguish parameter cases by production behavior rather than spelling matrices; removing that distinction fails while reformatting stays green.", + "timestamp": "2026-09-08T19:28:43.723Z" +} diff --git a/.2119/verdicts/REQ-003.2.2.json b/.2119/verdicts/REQ-003.2.2.json index b15f960..ebac493 100644 --- a/.2119/verdicts/REQ-003.2.2.json +++ b/.2119/verdicts/REQ-003.2.2.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-003.2.2--ba3a7f951d2b", + "reviewId": "REQ-003.2.2--8023b3f5d2ba", "requirementId": "REQ-003.2.2", - "hash": "ba3a7f951d2b", + "hash": "8023b3f5d2ba", "verdict": "pass", - "summary": "pass/fail write stable plain JSON after repairing final unignore rules, and init installs the same trackable verdict path", - "timestamp": "2026-08-03T17:05:00.760Z" + "summary": "writeVerdict emits plain JSON and pass, fail, and init repair ignore rules so .2119/verdicts records remain trackable", + "timestamp": "2026-09-08T20:11:02.943Z" } diff --git a/.2119/verdicts/REQ-003.4.2.json b/.2119/verdicts/REQ-003.4.2.json index 32291ef..078e676 100644 --- a/.2119/verdicts/REQ-003.4.2.json +++ b/.2119/verdicts/REQ-003.4.2.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-003.4.2--3e386c0febf7", + "reviewId": "REQ-003.4.2--3bdf40be1013", "requirementId": "REQ-003.4.2", - "hash": "3e386c0febf7", + "hash": "3bdf40be1013", "verdict": "pass", - "summary": "Risks section states the self-pass residual risk plainly and names the committed-verdict, hash-invalidation, and CI re-run mitigations", - "timestamp": "2026-08-08T09:06:52.659Z" + "summary": "README plainly states self-pass risk and the committed-verdict, hash-invalidation, and CI re-verification mitigations.", + "timestamp": "2026-09-08T19:28:06.904Z" } diff --git a/.2119/verdicts/REQ-003.6.3.json b/.2119/verdicts/REQ-003.6.3.json index 43acdc2..2083302 100644 --- a/.2119/verdicts/REQ-003.6.3.json +++ b/.2119/verdicts/REQ-003.6.3.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-003.6.3--4fa1d92e29a1", + "reviewId": "REQ-003.6.3--4d12d641b4cc", "requirementId": "REQ-003.6.3", - "hash": "4fa1d92e29a1", + "hash": "4d12d641b4cc", "verdict": "pass", - "summary": "Packaged CLI test rejects missing or mis-scoped audits, non-adversarial directives, stale passes, and any byte change to current, failing, orphan, or multiple verdict records", - "timestamp": "2026-08-03T16:15:20.573Z" + "summary": "The built CLI test proves --audit selects only current passing review IDs, emits counterexample-first instructions, and preserves current, failing, and orphan verdict bytes.", + "timestamp": "2026-09-08T20:18:56.701Z" } diff --git a/.2119/verdicts/REQ-003.8.1--7855d7409ca8.json b/.2119/verdicts/REQ-003.8.1--7855d7409ca8.json new file mode 100644 index 0000000..cd8f959 --- /dev/null +++ b/.2119/verdicts/REQ-003.8.1--7855d7409ca8.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.1--7855d7409ca8", + "requirementId": "REQ-003.8.1", + "hash": "7855d7409ca8", + "verdict": "pass", + "summary": "The 19 calibration fixtures each record a requirement, evidence, expected verdict, and reason, with known escapes, pass controls, and concrete cases for all five named review patterns.", + "timestamp": "2026-09-06T21:57:17.208Z" +} diff --git a/.2119/verdicts/REQ-003.8.2.json b/.2119/verdicts/REQ-003.8.2.json index c0ecc90..37ed182 100644 --- a/.2119/verdicts/REQ-003.8.2.json +++ b/.2119/verdicts/REQ-003.8.2.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-003.8.2--6e3c630dbed4", + "reviewId": "REQ-003.8.2--ecd4e37e46fe", "requirementId": "REQ-003.8.2", - "hash": "6e3c630dbed4", + "hash": "ecd4e37e46fe", "verdict": "pass", - "summary": "test-quality, direct-judgment, audit, and evidence-scoping instructions still expose every expected pass/fail distinction in corpus cases 001-014", - "timestamp": "2026-08-03T17:05:00.892Z" + "summary": "the template guidance covers every corpus escape and preserves the known-good delivered-surface, derived-inventory, and distinct-shape controls", + "timestamp": "2026-09-08T20:11:03.868Z" } diff --git a/.2119/verdicts/REQ-004.3.2--321edf5a8688.json b/.2119/verdicts/REQ-004.3.2--321edf5a8688.json deleted file mode 100644 index 649d893..0000000 --- a/.2119/verdicts/REQ-004.3.2--321edf5a8688.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "reviewId": "REQ-004.3.2--321edf5a8688", - "requirementId": "REQ-004.3.2", - "hash": "321edf5a8688", - "verdict": "pass", - "summary": "Runs real init on a pre-existing AGENTS.md and asserts begin/end markers appear exactly once (toHaveLength 1), pre-existing content preserved, plus a distinct phrase per mandated topic: spec-first, RFC2119 keyword, annotation example, fresh-context subagent, check-must-exit-0, draft critique, review --audit, different providers.", - "timestamp": "2026-07-10T22:58:12.253Z" -} diff --git a/.2119/verdicts/REQ-004.3.2.json b/.2119/verdicts/REQ-004.3.2.json new file mode 100644 index 0000000..61a3e5d --- /dev/null +++ b/.2119/verdicts/REQ-004.3.2.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.3.2--67fbf784f393", + "requirementId": "REQ-004.3.2", + "hash": "67fbf784f393", + "verdict": "pass", + "summary": "The CLI test extracts the single marker-delimited workflow section and checks every contracted instruction inside production-generated AGENTS.md output.", + "timestamp": "2026-09-08T19:31:43.956Z" +} diff --git a/.2119/verdicts/REQ-004.3.5--a7a677d01f1b.json b/.2119/verdicts/REQ-004.3.5--a7a677d01f1b.json deleted file mode 100644 index 410055b..0000000 --- a/.2119/verdicts/REQ-004.3.5--a7a677d01f1b.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "reviewId": "REQ-004.3.5--a7a677d01f1b", - "requirementId": "REQ-004.3.5", - "hash": "a7a677d01f1b", - "verdict": "pass", - "summary": "Same real-init test asserts the appended AGENTS.md body contains 'CI runs the same check', directly matching the requirement's criterion that the section state CI runs the same gate; absence would fail the assertion.", - "timestamp": "2026-07-10T22:58:13.875Z" -} diff --git a/.2119/verdicts/REQ-004.3.5.json b/.2119/verdicts/REQ-004.3.5.json new file mode 100644 index 0000000..2429ce2 --- /dev/null +++ b/.2119/verdicts/REQ-004.3.5.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.3.5--d07f18d7637a", + "requirementId": "REQ-004.3.5", + "hash": "d07f18d7637a", + "verdict": "pass", + "summary": "The CLI test now requires the CI-runs-the-same-check statement inside the generated marker-delimited AGENTS.md workflow section.", + "timestamp": "2026-09-08T19:31:44.142Z" +} diff --git a/.2119/verdicts/REQ-005.2.6.json b/.2119/verdicts/REQ-005.2.6.json index c2cce7c..b0ae321 100644 --- a/.2119/verdicts/REQ-005.2.6.json +++ b/.2119/verdicts/REQ-005.2.6.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-005.2.6--cf4818fa400c", + "reviewId": "REQ-005.2.6--e0116240b499", "requirementId": "REQ-005.2.6", - "hash": "cf4818fa400c", + "hash": "e0116240b499", "verdict": "pass", - "summary": "Verify-tag docs state arbitrary-shell execution and package.json-script trust equivalence beside the [verify] semantics", - "timestamp": "2026-08-08T09:06:58.759Z" + "summary": "README.md states that verify commands execute arbitrary shell from spec files and have the same trust level as package.json scripts.", + "timestamp": "2026-09-08T19:28:00.489Z" } diff --git a/.2119/verdicts/REQ-008.1.2.json b/.2119/verdicts/REQ-008.1.2.json index 392d17f..6622119 100644 --- a/.2119/verdicts/REQ-008.1.2.json +++ b/.2119/verdicts/REQ-008.1.2.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-008.1.2--4d6b2a51f53e", + "reviewId": "REQ-008.1.2--5b2c0227faf3", "requirementId": "REQ-008.1.2", - "hash": "4d6b2a51f53e", + "hash": "5b2c0227faf3", "verdict": "pass", - "summary": "README header block prominently states 2119 is not a test runner, CI replacement, or security boundary with the design.md link, before the adoption section", - "timestamp": "2026-08-08T09:07:04.379Z" + "summary": "README prominently names all three non-goals and links docs/design.md before the Use it in your repo adoption section.", + "timestamp": "2026-09-08T19:28:00.674Z" } diff --git a/.2119/verdicts/REQ-008.2.1.json b/.2119/verdicts/REQ-008.2.1.json index 649bc3f..1254134 100644 --- a/.2119/verdicts/REQ-008.2.1.json +++ b/.2119/verdicts/REQ-008.2.1.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-008.2.1--2cd0a760d9fd", + "reviewId": "REQ-008.2.1--19128a7cc3e9", "requirementId": "REQ-008.2.1", - "hash": "2cd0a760d9fd", + "hash": "19128a7cc3e9", "verdict": "pass", - "summary": "scaling.md covers all seven mandated topics with concrete recipes: pinning, separate CI gates, CODEOWNERS, independent-runner, explicit evidence globs, shared_evidence, and no-verify policy", - "timestamp": "2026-08-08T07:57:34.368Z" + "summary": "docs/scaling.md gives concrete guidance for all seven required safeguards, including separate test/check CI steps and --no-verify handling for untrusted contributions", + "timestamp": "2026-09-08T20:23:44.046Z" } diff --git a/.2119/verdicts/REQ-008.2.2.json b/.2119/verdicts/REQ-008.2.2.json index 7690ba1..0398297 100644 --- a/.2119/verdicts/REQ-008.2.2.json +++ b/.2119/verdicts/REQ-008.2.2.json @@ -1,8 +1,8 @@ { - "reviewId": "REQ-008.2.2--aa836f47efc5", + "reviewId": "REQ-008.2.2--9a3c36d2da8e", "requirementId": "REQ-008.2.2", - "hash": "aa836f47efc5", + "hash": "9a3c36d2da8e", "verdict": "pass", - "summary": "README's Diversify-then-audit bullet and scaling.md's cross-provider audit section both advise periodic sweeps plus targeted audits of challenging or high-consequence requirements", - "timestamp": "2026-08-08T09:07:17.187Z" + "summary": "README.md and docs/scaling.md each advise periodic review --audit sweeps with a different provider and individual audits for challenging or high-consequence requirements.", + "timestamp": "2026-09-08T20:23:42.473Z" } diff --git a/.2119/verdicts/REQ-008.2.3.json b/.2119/verdicts/REQ-008.2.3.json new file mode 100644 index 0000000..b3500e1 --- /dev/null +++ b/.2119/verdicts/REQ-008.2.3.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-008.2.3--a350fc47d58b", + "requirementId": "REQ-008.2.3", + "hash": "a350fc47d58b", + "verdict": "pass", + "summary": "README and scaling guide explain block-scoped churn avoidance, configured shared-helper invalidation, and guidance-only metric observation without automatic deletion", + "timestamp": "2026-09-08T20:23:56.328Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.1.json b/.2119/verdicts/self-supplied-evidence.1.1.json index f7914fd..913aab3 100644 --- a/.2119/verdicts/self-supplied-evidence.1.1.json +++ b/.2119/verdicts/self-supplied-evidence.1.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.1.1--afb9cc135da4", + "reviewId": "self-supplied-evidence.1.1--e260a01b5fde", "requirementId": "self-supplied-evidence.1.1", - "hash": "afb9cc135da4", + "hash": "e260a01b5fde", "verdict": "pass", - "summary": "The CLI-generated packet set is exhaustively exact-body checked and rejects omission or weakening of the concrete covering-test production-failure demand", - "timestamp": "2026-08-03T17:05:01.031Z" + "summary": "The production test-quality renderer requires naming the concrete production failure the covering test would catch, and the CLI routes pending reviews through that renderer.", + "timestamp": "2026-09-08T20:18:56.886Z" } diff --git a/.2119/verdicts/self-supplied-evidence.1.2.json b/.2119/verdicts/self-supplied-evidence.1.2.json index d8dd0cb..3b739e3 100644 --- a/.2119/verdicts/self-supplied-evidence.1.2.json +++ b/.2119/verdicts/self-supplied-evidence.1.2.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.1.2--c3c97de002d9", + "reviewId": "self-supplied-evidence.1.2--0cc78686b660", "requirementId": "self-supplied-evidence.1.2", - "hash": "c3c97de002d9", + "hash": "0cc78686b660", "verdict": "pass", - "summary": "Every CLI-generated test-quality packet is exact-body checked for file:line reachability independent of tests, fixtures, prompts, triggers, and decisive observations", - "timestamp": "2026-08-03T17:05:01.162Z" + "summary": "The production test-quality renderer requires file:line proof that production reaches the failure without fixtures or prompts supplying the trigger or decisive observation.", + "timestamp": "2026-09-08T20:18:57.123Z" } diff --git a/.2119/verdicts/self-supplied-evidence.2.1.json b/.2119/verdicts/self-supplied-evidence.2.1.json index ffd3296..05b4eee 100644 --- a/.2119/verdicts/self-supplied-evidence.2.1.json +++ b/.2119/verdicts/self-supplied-evidence.2.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.2.1--7fbcc5496008", + "reviewId": "self-supplied-evidence.2.1--59312c98f346", "requirementId": "self-supplied-evidence.2.1", - "hash": "7fbcc5496008", + "hash": "59312c98f346", "verdict": "pass", - "summary": "Every CLI-generated test-quality packet is exact-body checked for consumption of emitted values from separately invoked production components or data sources", - "timestamp": "2026-08-03T17:05:01.293Z" + "summary": "The production test-quality renderer defines this boundary as consuming a value emitted by a separately invoked production component or production data source.", + "timestamp": "2026-09-08T20:18:57.325Z" } diff --git a/.2119/verdicts/self-supplied-evidence.2.2.json b/.2119/verdicts/self-supplied-evidence.2.2.json index 58b5593..bad1dd8 100644 --- a/.2119/verdicts/self-supplied-evidence.2.2.json +++ b/.2119/verdicts/self-supplied-evidence.2.2.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.2.2--8796d4a6e793", + "reviewId": "self-supplied-evidence.2.2--038a254fd5af", "requirementId": "self-supplied-evidence.2.2", - "hash": "8796d4a6e793", + "hash": "038a254fd5af", "verdict": "pass", - "summary": "Built CLI dispatch output is read from generated files; whole-task equality rejects omission of file:line, conditional-boundary, or production-producer input provenance wording.", - "timestamp": "2026-08-03T17:05:06.235Z" + "summary": "The production test-quality renderer conditionally requires file:line proof that the covering test obtains input from the production producer.", + "timestamp": "2026-09-08T20:18:57.554Z" } diff --git a/.2119/verdicts/self-supplied-evidence.2.3.json b/.2119/verdicts/self-supplied-evidence.2.3.json index df7fae3..3a3614d 100644 --- a/.2119/verdicts/self-supplied-evidence.2.3.json +++ b/.2119/verdicts/self-supplied-evidence.2.3.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.2.3--3b0f7cd77ddc", + "reviewId": "self-supplied-evidence.2.3--ea931290f4a3", "requirementId": "self-supplied-evidence.2.3", - "hash": "3b0f7cd77ddc", + "hash": "ea931290f4a3", "verdict": "pass", - "summary": "Built CLI dispatch output is read without reshaping; whole-task equality rejects omission or weakening of the conditional production-shape evidence requirement.", - "timestamp": "2026-08-03T17:05:06.321Z" + "summary": "Test-quality judgment packets require file:line evidence that boundary input preserves the production producer's value shape.", + "timestamp": "2026-09-08T20:18:51.261Z" } diff --git a/.2119/verdicts/self-supplied-evidence.2.4.json b/.2119/verdicts/self-supplied-evidence.2.4.json index 38e223b..837b7a3 100644 --- a/.2119/verdicts/self-supplied-evidence.2.4.json +++ b/.2119/verdicts/self-supplied-evidence.2.4.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.2.4--9ea89fc088ad", + "reviewId": "self-supplied-evidence.2.4--8320faffe8dc", "requirementId": "self-supplied-evidence.2.4", - "hash": "9ea89fc088ad", + "hash": "8320faffe8dc", "verdict": "pass", - "summary": "Built CLI dispatch output is freshly generated and whole-task equality rejects loss of any initial/default/placeholder/sentinel case or new-versus-pre-existing distinction.", - "timestamp": "2026-08-03T17:05:06.404Z" + "summary": "Test-quality judgment packets require file:line evidence distinguishing a newly produced observation from an equal initial, default, placeholder, or sentinel value.", + "timestamp": "2026-09-08T20:18:51.446Z" } diff --git a/.2119/verdicts/self-supplied-evidence.3.1.json b/.2119/verdicts/self-supplied-evidence.3.1.json index db5103b..ea746a2 100644 --- a/.2119/verdicts/self-supplied-evidence.3.1.json +++ b/.2119/verdicts/self-supplied-evidence.3.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.3.1--16a37e67b6e3", + "reviewId": "self-supplied-evidence.3.1--d2069bea64d7", "requirementId": "self-supplied-evidence.3.1", - "hash": "16a37e67b6e3", + "hash": "d2069bea64d7", "verdict": "pass", - "summary": "Built CLI dispatch output is exercised; whole-task equality rejects definitions omitting binary/service, external invocation, or the gate-process boundary.", - "timestamp": "2026-08-03T17:05:06.489Z" + "summary": "Test-quality judgment packets define the gate/runtime-environment boundary as invoking a binary or service outside the gate process.", + "timestamp": "2026-09-08T20:18:51.682Z" } diff --git a/.2119/verdicts/self-supplied-evidence.3.2.json b/.2119/verdicts/self-supplied-evidence.3.2.json index 4dabd20..6f96a4c 100644 --- a/.2119/verdicts/self-supplied-evidence.3.2.json +++ b/.2119/verdicts/self-supplied-evidence.3.2.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.3.2--e873b585f1a6", + "reviewId": "self-supplied-evidence.3.2--111d23ee4498", "requirementId": "self-supplied-evidence.3.2", - "hash": "e873b585f1a6", + "hash": "111d23ee4498", "verdict": "pass", - "summary": "Built CLI dispatch output is exercised; whole-task equality rejects omission of the conditional boundary, production provisioning declaration, absent-dependency failure path, or either half of both.", - "timestamp": "2026-08-03T17:05:06.568Z" + "summary": "Test-quality judgment packets conditionally require file:line evidence for both production provisioning and the production absence-failure path.", + "timestamp": "2026-09-08T20:18:51.902Z" } diff --git a/.2119/verdicts/self-supplied-evidence.4.1.json b/.2119/verdicts/self-supplied-evidence.4.1.json index 64873e2..91a50fd 100644 --- a/.2119/verdicts/self-supplied-evidence.4.1.json +++ b/.2119/verdicts/self-supplied-evidence.4.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.4.1--c246036538fc", + "reviewId": "self-supplied-evidence.4.1--1ffea4686587", "requirementId": "self-supplied-evidence.4.1", - "hash": "c246036538fc", + "hash": "1ffea4686587", "verdict": "pass", - "summary": "Generated test-quality packets from the spawned production CLI retain the explicit fail-on-absent-or-self-supplied-provenance direction for every parsed target.", - "timestamp": "2026-08-03T17:05:04.663Z" + "summary": "The test-quality instruction branch explicitly records FAIL when applicable provenance evidence is absent or production cannot produce the failure independently of test setup.", + "timestamp": "2026-09-08T20:18:51.584Z" } diff --git a/.2119/verdicts/self-supplied-evidence.5.1.json b/.2119/verdicts/self-supplied-evidence.5.1.json index 0f9af7f..275a976 100644 --- a/.2119/verdicts/self-supplied-evidence.5.1.json +++ b/.2119/verdicts/self-supplied-evidence.5.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.5.1--c1a086a4ad44", + "reviewId": "self-supplied-evidence.5.1--d060740a67b8", "requirementId": "self-supplied-evidence.5.1", - "hash": "c1a086a4ad44", + "hash": "d060740a67b8", "verdict": "pass", - "summary": "Generated direct-judgment packets for parsed [review] targets exactly retain the direct task and omit the test-quality provenance section.", - "timestamp": "2026-08-03T17:05:04.793Z" + "summary": "The direct-judgment requirement branch omits the production-provenance questions that are emitted only by the test-quality branch.", + "timestamp": "2026-09-08T20:18:51.802Z" } diff --git a/.2119/verdicts/self-supplied-evidence.6.1.json b/.2119/verdicts/self-supplied-evidence.6.1.json index b5c2350..7044d86 100644 --- a/.2119/verdicts/self-supplied-evidence.6.1.json +++ b/.2119/verdicts/self-supplied-evidence.6.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.6.1--9eff2a724166", + "reviewId": "self-supplied-evidence.6.1--966521ec57f3", "requirementId": "self-supplied-evidence.6.1", - "hash": "9eff2a724166", + "hash": "966521ec57f3", "verdict": "pass", - "summary": "Built lint accepts isolated string/object/array/template/regexp/number/bigint/boolean/null inputs and named/static/imported/arrow factory forms without provenance violations.", - "timestamp": "2026-08-03T17:07:21.073Z" + "summary": "The real lint CLI accepts scalar, object, array, template, regexp, and primitive literals plus named, static, constructor, instance-method, imported, and arrow factory cases.", + "timestamp": "2026-09-08T20:18:52.020Z" } diff --git a/.2119/verdicts/self-supplied-evidence.7.1.json b/.2119/verdicts/self-supplied-evidence.7.1.json index 8368652..6ba1b2e 100644 --- a/.2119/verdicts/self-supplied-evidence.7.1.json +++ b/.2119/verdicts/self-supplied-evidence.7.1.json @@ -1,8 +1,8 @@ { - "reviewId": "self-supplied-evidence.7.1--22a18023494c", + "reviewId": "self-supplied-evidence.7.1--baf1b43a6c2e", "requirementId": "self-supplied-evidence.7.1", - "hash": "22a18023494c", + "hash": "baf1b43a6c2e", "verdict": "pass", - "summary": "Every generated standard packet and every producible audit packet retains final tool-authored guidance preserving concrete member names and singular/plural scope; operator custom criteria are separately delimited before the task.", - "timestamp": "2026-08-03T17:05:05.044Z" + "summary": "Standard and audit instruction renderers each require verdict summaries to preserve cited member names and singular or plural scope.", + "timestamp": "2026-09-08T20:18:52.244Z" } diff --git a/AGENTS.md b/AGENTS.md index 8e20de7..39cbf8d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -3,6 +3,7 @@ This repository enforces spec-driven testing with [2119](https://www.rfc-editor.org/rfc/rfc2119). + **When planning a feature**, write or update a spec in `specs/` first. Every requirement is a numbered item under a `### REQ-NNN.M` heading with exactly one RFC 2119 keyword, stating an observable outcome — not an implementation @@ -11,6 +12,15 @@ against a new spec**, dispatch a fresh-context reviewer to critique the draft requirements themselves: outcome-stated, individually testable, one obligation each. A flawed requirement steers the whole implementation wrong. + +**Classify before making it normative**: use requirements for observable product +behavior and narrowly scoped text actually delivered through a product surface. +Keep test strategy, CI commands, review procedure, migration bookkeeping, and +implementation notes non-normative unless they are themselves supported +interfaces. Consolidate duplicate requirements instead of splitting wording into +proof obligations. + + **Requirement granularity**: A first-pass feature spec should aim for around 3–8 enforced `MUST` requirements. Prefer workflow-level requirements (what the user can observably do) over implementation-step requirements (how the code @@ -22,19 +32,24 @@ instead of describing product behavior. Use `SHOULD` for polish and edge cases, `[manual]` for UI-only behaviors, and notes or acceptance-checklist bullets for implementation details rather than making every detail an enforced `MUST`. + **When implementing**, every MUST/SHALL requirement needs at least one test annotated with a comment containing its ID, e.g. `// 2119: REQ-001.2.3` (the marker line must start with a comment leader). Write tests that would genuinely fail if the requirement were violated — including its negative space: what the requirement forbids needs a rejection test, not just what it allows. A fresh-context reviewer judges each test's honesty; tautological or over-mocked -tests will be rejected. +tests will be rejected. One behavioral test may cover multiple requirement IDs: +add annotations or cross-references when the evidence is already sufficient, +rather than duplicating the test for traceability. + **Reviewer diversity**: use reviewer models from different providers, routinely or as periodic `npx rfc2119 review --audit` sweeps — adversarial audits of passing verdicts. Audit especially the challenging or high-consequence requirements; a single model family shares blind spots. + **Before finishing any task**, run `npx rfc2119 check`. It must exit 0. If it reports pending judgment reviews, run `npx rfc2119 review --dispatch` and dispatch each instruction file in `.2119/reviews/` to a fresh-context subagent diff --git a/README.md b/README.md index 094c063..70a752f 100644 --- a/README.md +++ b/README.md @@ -97,7 +97,9 @@ The design splits enforcement by what each layer can actually guarantee: requirement — and the old verdict silently stops counting; edit an *unrelated* test in the same file and it doesn't. `2119 pass` refuses IDs whose hash doesn't match current content, so verdicts can't be pre-computed - or replayed. + or replayed. Configure `shared_evidence` only for fixtures or helpers that can + genuinely neutralize many tests: those files deliberately invalidate every + dependent verdict when edited, trading broader churn for helper integrity. - **Verdicts are committed and schema-validated.** `.2119/verdicts/*.json` files carry the verdict, summary, and timestamp, so every review decision shows up in the PR diff for humans to audit. The gate counts a verdict only diff --git a/docs/scaling.md b/docs/scaling.md index ff55379..f560879 100644 --- a/docs/scaling.md +++ b/docs/scaling.md @@ -90,6 +90,21 @@ shared_evidence: Their content then joins every test-quality hash. The cost is honest churn: editing a shared helper re-opens every dependent review, which is exactly what should happen. +Ordinary test-quality verdicts instead hash the annotated evidence block and the file prelude. +This keeps an edit to an unrelated neighboring annotation block from invalidating evidence it +cannot affect. Renaming the evidence file or changing names, formatting, or setup inside the covered +block or its prelude still invalidates the verdict. The two scopes are a deliberate tradeoff: narrow +blocks avoid unrelated churn; explicitly configured shared evidence preserves integrity where a +common mock or helper could neutralize many tests. + +## Anti-accretion rollout + +Treat requirement counts, execution-to-requirement ratios, matrix sizes, and verdict invalidation +rates as review prompts or trends, never pass/fail proxies for evidence quality. Roll out +anti-accretion through author and reviewer guidance first. The tool does not automatically delete, +invalidate, or rewrite a repository's requirements or tests; maintainers decide what evidence is +redundant after reviewing behavior. + ## Periodic adversarial audits (cross-provider) Fresh context is not fresh framing: reviewers sharing one model family and one instruction diff --git a/eval/calibration/015-prose-tripwire.md b/eval/calibration/015-prose-tripwire.md new file mode 100644 index 0000000..20ea078 --- /dev/null +++ b/eval/calibration/015-prose-tripwire.md @@ -0,0 +1,22 @@ +--- +expected_verdict: fail +prompt: test-quality +source: issue #38 +failure_mode: prose tripwire mistaken for behavioral evidence +--- + +## Requirement + +> The generated help MUST tell users how to select an output format. + +## Evidence + +```ts +// 2119: FIX-001.1.1 +expect(readFileSync("specs/FIX-001-help.md", "utf8")).toContain("select an output format"); +``` + +## Why the correct verdict is FAIL + +The assertion reads the specification, not the delivered help surface. A wording-only edit breaks +it while removing the product behavior does not. diff --git a/eval/calibration/016-redundant-parameter-matrix.md b/eval/calibration/016-redundant-parameter-matrix.md new file mode 100644 index 0000000..793bb06 --- /dev/null +++ b/eval/calibration/016-redundant-parameter-matrix.md @@ -0,0 +1,24 @@ +--- +expected_verdict: fail +prompt: test-quality +source: issue #38 +failure_mode: parameter spellings multiply one production behavior +--- + +## Requirement + +> Invalid color values MUST be rejected. + +## Evidence + +```ts +// 2119: FIX-001.1.1 +it.each(["wat", "nope", "invalid", "unknown", "???"])("%s is rejected", (value) => { + expect(parseColor(value)).toEqual({ ok: false }); +}); +``` + +## Why the correct verdict is FAIL + +Every value is the same unrecognized-token shape and reaches the same branch. The matrix adds +executions without evidence for distinct malformed shapes. diff --git a/eval/calibration/017-derived-security-inventory.md b/eval/calibration/017-derived-security-inventory.md new file mode 100644 index 0000000..f275eb4 --- /dev/null +++ b/eval/calibration/017-derived-security-inventory.md @@ -0,0 +1,24 @@ +--- +expected_verdict: pass +prompt: test-quality +source: issue #38 +failure_mode: legitimate derived inventory mistaken for a frozen literal +--- + +## Requirement + +> Every registered administrative route MUST enforce authentication. + +## Evidence + +```ts +// 2119: FIX-001.1.1 +const routes = runningApp.registeredRoutes().filter((route) => route.admin); +expect(routes.length).toBeGreaterThan(0); +for (const route of routes) expect(requestWithoutAuth(route)).toHaveStatus(401); +``` + +## Why the correct verdict is PASS + +The inventory comes from the running product rather than a frozen test literal, and the +zero-subject assertion prevents vacuous success. diff --git a/eval/calibration/018-delivered-agent-instruction.md b/eval/calibration/018-delivered-agent-instruction.md new file mode 100644 index 0000000..5d2da02 --- /dev/null +++ b/eval/calibration/018-delivered-agent-instruction.md @@ -0,0 +1,23 @@ +--- +expected_verdict: pass +prompt: test-quality +source: issue #38 +failure_mode: delivered instruction mistaken for irrelevant prose pinning +--- + +## Requirement + +> The generated agent workflow MUST instruct the agent to run the final check. + +## Evidence + +```ts +// 2119: FIX-001.1.1 +init(fixture); +expect(readFileSync(join(fixture, "AGENTS.md"), "utf8")).toContain("run `npx rfc2119 check`"); +``` + +## Why the correct verdict is PASS + +`AGENTS.md` is the delivered product surface that controls agent behavior. The assertion pins only +the required instruction, not its heading order or unrelated prose. diff --git a/eval/calibration/019-distinct-malformed-shapes.md b/eval/calibration/019-distinct-malformed-shapes.md new file mode 100644 index 0000000..950380e --- /dev/null +++ b/eval/calibration/019-distinct-malformed-shapes.md @@ -0,0 +1,26 @@ +--- +expected_verdict: pass +prompt: test-quality +source: issue #38 +failure_mode: meaningful negative-space table mistaken for a redundant matrix +--- + +## Requirement + +> Malformed envelopes MUST be rejected before dispatch. + +## Evidence + +```ts +// 2119: FIX-001.1.1 +it.each([ + ["missing kind", { payload: {} }], + ["wrong payload type", { kind: "run", payload: 1 }], + ["unknown kind", { kind: "erase", payload: {} }], +])("%s", (_name, envelope) => expect(dispatch(envelope)).toEqual({ ok: false })); +``` + +## Why the correct verdict is PASS + +The cases exercise distinct schema and dispatch boundaries rather than alternate spellings of one +invalid token. diff --git a/specs/REQ-003-judgment-reviews.md b/specs/REQ-003-judgment-reviews.md index 682b3c0..8f60eb3 100644 --- a/specs/REQ-003-judgment-reviews.md +++ b/specs/REQ-003-judgment-reviews.md @@ -29,8 +29,12 @@ their findings (not empty markers), and they live in version control. 7. An annotation's evidence block MUST comprise the file's prelude (all content before the file's first annotation, hashed once per file) plus the text from the annotation's line through the line before the file's next annotation or the end of file, so shared imports and mocks stay under the hash while unrelated tests fall outside it. 8. When the optional `shared_evidence` config key lists globs, the content of every matching file MUST be included in the hash input of every test-quality review, so shared fixtures and helper modules cannot change without invalidating the verdicts that depend on them. 9. A `[review: ]` tag whose globs match no files MUST produce a check violation naming the requirement and the unmatched globs, rather than silently degrading to a text-only hash. -10. Instruction files MUST direct the reviewer to enumerate the requirement's conjuncts and boundary terms (words like `comment`, `exactly`, `only`, `begins with`) and, for each, construct the nearest violating input and confirm a test rejects it — a review that cannot name a rejected counterexample for a boundary term is not a pass. +10. Instruction files MUST direct the reviewer to name one concrete implementation change that violates the requirement and confirm the cited evidence would detect it, without demanding a counterexample for every word or a Cartesian product of equivalent inputs. 11. Instruction files MUST direct the reviewer to fail with a finding when the requirement itself is ambiguous, untestable, or states an implementation mechanism rather than an observable outcome, since a bad requirement honestly tested is still a bad requirement. +12. Instruction files MUST direct the reviewer to name one legitimate change that preserves the requirement's meaning and confirm the cited evidence would remain green. +13. Instruction files MUST permit one evidence body to cover multiple requirement IDs and direct reviewers to request an annotation or cross-reference, not a duplicate test, when existing evidence already detects the violating change. +14. Instruction files MUST direct the reviewer to reject evidence whose only value is pinning irrelevant wording, layout, digests, or implementation organization, while preserving legitimate delivered-text contracts, derived real-product inventories that fail loudly on zero subjects, and maintainable snapshots. +15. Instruction files MUST direct the reviewer to judge parameter cases by whether they exercise meaningfully distinct production behavior, not by whether universal wording can generate more spellings or combinations. ### REQ-003.2: Verdict recording @@ -88,5 +92,5 @@ changes ratchet: they may never lose the ability to catch a past escape. The cor committed fixtures; optimization loops over it (e.g. SkillOpt-style tuning) are dev-time experiments outside this tool, whose proposed edits land through normal spec amendments. -1. The repository MUST maintain a calibration corpus under `eval/calibration/` of fixture cases — each a requirement, its evidence, the expected verdict, and the reason — including a case for every known review escape. [review: eval/calibration/**] +1. The repository MUST maintain a calibration corpus under `eval/calibration/` of fixture cases — each a requirement, its evidence, the expected verdict, and the reason — including known review escapes and controls for prose tripwires, redundant parameter matrices, derived security inventories, delivered agent instructions, and distinct malformed data shapes. [review: eval/calibration/**] 2. Changes to the reviewer instruction template MUST preserve detection of every corpus case: a template revision under which a reviewer following the instructions would no longer catch a corpus escape is a regression, not a simplification. [review: eval/calibration/**, src/review.ts] diff --git a/specs/REQ-004-agent-integration.md b/specs/REQ-004-agent-integration.md index 3b3c486..3af0706 100644 --- a/specs/REQ-004-agent-integration.md +++ b/specs/REQ-004-agent-integration.md @@ -46,7 +46,7 @@ plugins and are planned as thin native packages. ### REQ-004.3: Universal fallback layer 1. `2119 init` MUST create a commented `.2119.yml` and a template spec when none exist. -2. `2119 init` MUST append a marker-delimited workflow section to `AGENTS.md` describing spec-first planning, draft-time spec critique (dispatch a fresh reviewer to critique new requirements — outcome-stated, individually testable, one obligation each — before writing tests), test annotations, judgment reviews, reviewer-model diversity (use models from different providers routinely or as periodic `review --audit` sweeps, especially for challenging or high-consequence requirements), and the `2119 check` gate, exactly once. +2. `2119 init` MUST append exactly once a marker-delimited workflow section to `AGENTS.md` that describes spec-first planning; distinguishes observable product behavior and narrowly scoped delivered-text contracts from non-normative verification or maintenance notes; asks authors to consolidate duplicate requirements and annotate shared behavioral evidence instead of duplicating tests; requires draft-time fresh-context spec critique; explains test annotations and judgment reviews; recommends reviewer-model diversity; and names the `2119 check` gate, with the section's key instruction-bearing phrases forming part of the delivered-text contract. 3. `2119 init --git-hook` MUST install a git pre-commit hook that runs `2119 check`, refusing to overwrite an existing pre-commit hook it did not create. 4. `2119 init --ci` MUST write a GitHub Actions workflow that runs `2119 check` on pull requests. 5. The AGENTS.md section MUST state that CI runs the same check, so agents on hookless platforms know the gate cannot be skipped. diff --git a/specs/REQ-008-honest-boundaries.md b/specs/REQ-008-honest-boundaries.md index b902fb5..b6b5f87 100644 --- a/specs/REQ-008-honest-boundaries.md +++ b/specs/REQ-008-honest-boundaries.md @@ -22,3 +22,4 @@ escape hatches and documented recipes, never additions to the default path. 1. The repository MUST contain `docs/scaling.md` covering at minimum: exact-version pinning, running the project test suite alongside `2119 check` as separate CI gates, CODEOWNERS protection for specs and verdicts, the independent-runner reviewer recipe, explicit evidence globs rather than bare `[review]` tags for critical requirements, `shared_evidence` for shared fixtures, and a `[verify]` policy for repositories accepting untrusted contributions. [review: docs/scaling.md] 2. The documentation MUST advise periodic cross-provider `review --audit` sweeps as part of quality assurance, and targeted audits for particularly challenging or high-consequence requirements, in both the README and the scaling guide. [review: README.md, docs/scaling.md] +3. The documentation MUST explain that annotation-block hashing avoids unrelated verdict churn while configured shared evidence deliberately invalidates dependent verdicts to preserve helper integrity, and that anti-accretion rollout is guidance and observational reporting rather than automatic test deletion or metric-based gating. [review: README.md, docs/scaling.md] diff --git a/specs/self-supplied-evidence.md b/specs/self-supplied-evidence.md index ff7a72c..dab7365 100644 --- a/specs/self-supplied-evidence.md +++ b/specs/self-supplied-evidence.md @@ -79,39 +79,37 @@ inside the behavior under test is not enough. A runtime-environment boundary exi invokes a binary or service outside its own process. These definitions trigger focused provenance traces without making every pure unit test pay the cost. -This feature's own acceptance tests invoke the built CLI's real `review --dispatch` workflow -against this checked-in file-scoped spec and its annotated tests, then inspect instructions that -workflow actually writes. They do not satisfy coverage by directly calling the instruction -renderer with a hand-built requirement or by asserting against a separately copied prompt fixture. -Thus the production parser, coverage resolver, review-target computation, and instruction writer -supply the artifacts under assertion. +These delivered-text obligations use direct judgment review because deterministic prose matching +would either admit negated lookalikes or reject legitimate paraphrases. The declared evidence spans +the production CLI path and instruction renderer; the reviewer judges the emitted contract from +those sources instead of accepting a copied prompt fixture or a wording snapshot. ## Requirements ### 1: Independent production failure -1. Each generated test-quality review instruction MUST require the reviewer to name a concrete production failure the covering test would catch. -2. Each generated test-quality review instruction MUST require `file:line` evidence that production can reach the named failure without the test, its fixtures, or its prompts supplying the triggering input or decisive observation. +1. Each generated test-quality review instruction MUST require the reviewer to name a concrete production failure the covering test would catch. [review: src/cli.ts, src/review.ts] +2. Each generated test-quality review instruction MUST require `file:line` evidence that production can reach the named failure without the test, its fixtures, or its prompts supplying the triggering input or decisive observation. [review: src/cli.ts, src/review.ts] ### 2: Conditional boundary provenance -1. Each generated test-quality review instruction MUST define a producer/consumer boundary as consumption of a value emitted by a separately invoked production component or production data source. -2. Each generated test-quality review instruction MUST require the reviewer, whenever a producer/consumer boundary exists, to cite `file:line` evidence that the covering test obtains its input through that production producer. -3. Each generated test-quality review instruction MUST require the reviewer, whenever a producer/consumer boundary exists, to cite `file:line` evidence that the input exercised by the covering test preserves the production producer's value shape. -4. Each generated test-quality review instruction MUST require the reviewer, whenever the decisive observation can equal an initial, default, placeholder, or sentinel value, to cite `file:line` evidence that the covering test distinguishes a newly produced observation from that pre-existing value. +1. Each generated test-quality review instruction MUST define a producer/consumer boundary as consumption of a value emitted by a separately invoked production component or production data source. [review: src/cli.ts, src/review.ts] +2. Each generated test-quality review instruction MUST require the reviewer, whenever a producer/consumer boundary exists, to cite `file:line` evidence that the covering test obtains its input through that production producer. [review: src/cli.ts, src/review.ts] +3. Each generated test-quality review instruction MUST require the reviewer, whenever a producer/consumer boundary exists, to cite `file:line` evidence that the input exercised by the covering test preserves the production producer's value shape. [review: src/cli.ts, src/review.ts] +4. Each generated test-quality review instruction MUST require the reviewer, whenever the decisive observation can equal an initial, default, placeholder, or sentinel value, to cite `file:line` evidence that the covering test distinguishes a newly produced observation from that pre-existing value. [review: src/cli.ts, src/review.ts] ### 3: Declared runtime environment -1. Each generated test-quality review instruction MUST define a gate/runtime-environment boundary as invocation of a binary or service outside the gate's own process. -2. Each generated test-quality review instruction MUST require the reviewer, whenever a gate/runtime-environment boundary exists, to cite `file:line` evidence of both the dependency's production provisioning declaration and the production path that fails when the dependency is absent. +1. Each generated test-quality review instruction MUST define a gate/runtime-environment boundary as invocation of a binary or service outside the gate's own process. [review: src/cli.ts, src/review.ts] +2. Each generated test-quality review instruction MUST require the reviewer, whenever a gate/runtime-environment boundary exists, to cite `file:line` evidence of both the dependency's production provisioning declaration and the production path that fails when the dependency is absent. [review: src/cli.ts, src/review.ts] ### 4: Decidable verdicts -1. Each generated test-quality review instruction MUST direct the reviewer to fail the judgment when any provenance evidence required by its applicable questions is absent or shows that production cannot produce the claimed failure independently of the test setup. +1. Each generated test-quality review instruction MUST direct the reviewer to fail the judgment when any provenance evidence required by its applicable questions is absent or shows that production cannot produce the claimed failure independently of the test setup. [review: src/cli.ts, src/review.ts] ### 5: Scope -1. Generated direct-judgment instructions for `[review]` requirements MUST remain exempt from the test-quality provenance questions. +1. Generated direct-judgment instructions for `[review]` requirements MUST remain exempt from the test-quality provenance questions. [review: src/cli.ts, src/review.ts] ### 6: Lint compatibility @@ -119,4 +117,4 @@ supply the artifacts under assertion. ### 7: Evidence-bounded verdict wording -1. Each generated review instruction MUST direct the reviewer to keep the verdict summary's subject no broader than the cited evidence, preserving concrete member names and singular or plural scope instead of promoting member-specific evidence into a category claim. +1. Each generated review instruction MUST direct the reviewer to keep the verdict summary's subject no broader than the cited evidence, preserving concrete member names and singular or plural scope instead of promoting member-specific evidence into a category claim. [review: src/cli.ts, src/review.ts] diff --git a/src/init.ts b/src/init.ts index 5ed9846..189dfb1 100644 --- a/src/init.ts +++ b/src/init.ts @@ -62,20 +62,21 @@ requirements (RFC 2119 §7). 2. The system SHOULD . `; -export const AGENTS_MD_SECTION = ` -## Requirements workflow (2119) - -This repository enforces spec-driven testing with [2119](https://www.rfc-editor.org/rfc/rfc2119). - -**When planning a feature**, write or update a spec in \`specs/\` first. Every +export const AGENT_WORKFLOW_TOPICS = { + planning: `**When planning a feature**, write or update a spec in \`specs/\` first. Every requirement is a numbered item under a \`### REQ-NNN.M\` heading with exactly one RFC 2119 keyword, stating an observable outcome — not an implementation mechanism. Run \`npx rfc2119 lint\` after editing specs. **Before writing tests against a new spec**, dispatch a fresh-context reviewer to critique the draft requirements themselves: outcome-stated, individually testable, one obligation -each. A flawed requirement steers the whole implementation wrong. - -**Requirement granularity**: A first-pass feature spec should aim for around +each. A flawed requirement steers the whole implementation wrong.`, + classification: `**Classify before making it normative**: use requirements for observable product +behavior and narrowly scoped text actually delivered through a product surface. +Keep test strategy, CI commands, review procedure, migration bookkeeping, and +implementation notes non-normative unless they are themselves supported +interfaces. Consolidate duplicate requirements instead of splitting wording into +proof obligations.`, + granularity: `**Requirement granularity**: A first-pass feature spec should aim for around 3–8 enforced \`MUST\` requirements. Prefer workflow-level requirements (what the user can observably do) over implementation-step requirements (how the code achieves it). Spec sizing smells — reconsider the spec if: one feature produces @@ -84,26 +85,35 @@ restate internal steps rather than user-visible outcomes; a single test would cover many requirements at once; or requirements say "MUST cover" or "MUST test" instead of describing product behavior. Use \`SHOULD\` for polish and edge cases, \`[manual]\` for UI-only behaviors, and notes or acceptance-checklist bullets for -implementation details rather than making every detail an enforced \`MUST\`. - -**When implementing**, every MUST/SHALL requirement needs at least one test +implementation details rather than making every detail an enforced \`MUST\`.`, + implementation: `**When implementing**, every MUST/SHALL requirement needs at least one test annotated with a comment containing its ID, e.g. \`// 2119: REQ-001.2.3\` (the marker line must start with a comment leader). Write tests that would genuinely fail if the requirement were violated — including its negative space: what the requirement forbids needs a rejection test, not just what it allows. A fresh-context reviewer judges each test's honesty; tautological or over-mocked -tests will be rejected. - -**Reviewer diversity**: use reviewer models from different providers, routinely +tests will be rejected. One behavioral test may cover multiple requirement IDs: +add annotations or cross-references when the evidence is already sufficient, +rather than duplicating the test for traceability.`, + reviewerDiversity: `**Reviewer diversity**: use reviewer models from different providers, routinely or as periodic \`npx rfc2119 review --audit\` sweeps — adversarial audits of passing verdicts. Audit especially the challenging or high-consequence -requirements; a single model family shares blind spots. - -**Before finishing any task**, run \`npx rfc2119 check\`. It must exit 0. If it +requirements; a single model family shares blind spots.`, + gate: `**Before finishing any task**, run \`npx rfc2119 check\`. It must exit 0. If it reports pending judgment reviews, run \`npx rfc2119 review --dispatch\` and dispatch each instruction file in \`.2119/reviews/\` to a fresh-context subagent (never review your own work in the same context). CI runs the same check, so -skipping it locally only defers the failure. +skipping it locally only defers the failure.`, +} as const; + +export const AGENTS_MD_SECTION = ` +## Requirements workflow (2119) + +This repository enforces spec-driven testing with [2119](https://www.rfc-editor.org/rfc/rfc2119). + +${Object.entries(AGENT_WORKFLOW_TOPICS) + .map(([id, text]) => `\n${text}`) + .join("\n\n")} `; diff --git a/src/review.ts b/src/review.ts index bd74e1b..c2dfcb4 100644 --- a/src/review.ts +++ b/src/review.ts @@ -11,6 +11,46 @@ import type { VerdictFile } from "./verdict.js"; export const REVIEWS_DIR = ".2119/reviews"; +export const TEST_QUALITY_GUIDANCE = [ + { + id: "violating-change", + text: `Name one concrete implementation change that violates the requirement and confirm the cited +evidence would fail. Before selecting it, scan the requirement's conjuncts, boundaries, precedence +rules, grammar shapes, and distinct data shapes; choose the probe most likely to expose uncovered +behavior. When a requirement quantifies over a set or names a defined grammar, confirm the evidence +exercises boundary members or edge productions, not just the easiest member. Do not demand a +counterexample for every word or a Cartesian product of inputs that exercise the same production +behavior.`, + }, + { + id: "legitimate-change", + text: `Name one legitimate change that preserves the requirement's meaning — such as paraphrasing, +renaming, reformatting, adding a sibling item, or reorganizing files — and confirm the cited +evidence would stay green.`, + }, + { + id: "shared-evidence", + text: `One evidence body may cover multiple requirement IDs. When existing evidence already rejects +the violating change, request an annotation or explicit cross-reference, not a duplicate test.`, + }, + { + id: "irrelevant-pins", + text: `Reject evidence whose only value is pinning irrelevant wording, layout, digests, or +implementation organization. Preserve legitimate contracts for text delivered as the product +surface, inventories derived from the real product that fail loudly on zero subjects, and snapshots +with an explicit, inexpensive update path.`, + }, + { + id: "parameter-cases", + text: `For parameterized evidence, ask whether each value exercises meaningfully distinct production +behavior. Universal wording alone is not a reason to demand every spelling or combination.`, + }, +] as const; + +export const REQUIREMENT_QUALITY_GUIDANCE = `If the requirement itself is ambiguous, untestable, or +states an implementation mechanism rather than an observable outcome, fail with that finding — a +bad requirement honestly tested is still a bad requirement.`; + export interface ReviewTask { reviewId: string; requirement: Requirement; @@ -203,10 +243,11 @@ ${evidenceList} ## Your task **Construct a concrete mutant or input under which this requirement is violated while every -covering test stays green.** Enumerate the requirement's conjuncts and boundary terms; probe the -negative space (what must be refused, not what is accepted); consider shared fixtures, preludes, -and paths the tests never touch. Reason from the requirement's text, never from the -implementation's current behavior. +covering test stays green.** Probe the negative space (what must be refused, not what is accepted); +consider shared fixtures, preludes, and paths the tests never touch. Before selecting a mutant, +scan every conjunct, boundary, precedence rule, grammar shape, and distinct data shape. Prefer a +discriminating counterexample over exhaustive permutations of equivalent inputs. Reason from the +requirement's text, never from the implementation's current behavior. - If you find such a counterexample: record a FAIL with the mutant described concretely enough to reproduce. @@ -279,17 +320,19 @@ Read the requirement and each evidence file's tests annotated with \`2119: ${t.r Record FAIL when applicable provenance evidence is absent or shows that production cannot produce the failure independently of the test setup. -**Counterexample obligation:** enumerate the requirement's conjuncts and boundary terms (words -like "comment", "exactly", "only", "begins with"). For each, construct the nearest violating -input — the almost-conforming case the requirement forbids — and confirm a test rejects it. -When a requirement names a grammar or other defined input language, enumerate and probe its edge -productions rather than accepting coverage of only the most common form. -Do not reason from the implementation's current behavior; reason from the requirement's text. -A review that cannot name a rejected counterexample for a boundary term is not a pass.` +**Symmetric change probes (a PASS is forbidden without both):** + +${TEST_QUALITY_GUIDANCE.map((item) => `\n${item.text}`).join("\n\n")} + +Do not reason from the implementation's current behavior; reason from the requirement's text.` : `**Is this requirement genuinely satisfied by the current state of the evidence files?** Read the requirement and the evidence files and judge compliance directly. This requirement was tagged \`[review]\` because it needs judgment rather than a test.`; + const symmetricGuidance = `**Symmetric change probes (a PASS is forbidden without both):** + +${TEST_QUALITY_GUIDANCE.map((item) => `\n${item.text}`).join("\n\n")}`; + // Judgment-heavy [review]-tagged requirements warrant the dispatcher's own // (typically stronger) model; routine test-quality reviews suit the pinned // cheaper tier (REQ-003.5.2, REQ-003.5.3). Multiple configured models mean @@ -328,11 +371,10 @@ ${custom.content} ## Your task -${question} +${question}${t.kind === "requirement" ? `\n\n${symmetricGuidance}` : ""} -**Judge the requirement too:** if the requirement itself is ambiguous, untestable, or states an -implementation mechanism rather than an observable outcome, fail with that finding — a bad -requirement honestly tested is still a bad requirement. + +**Judge the requirement too:** ${REQUIREMENT_QUALITY_GUIDANCE} ## Recording your verdict diff --git a/tests/cli.test.ts b/tests/cli.test.ts index 9cff497..279f787 100644 --- a/tests/cli.test.ts +++ b/tests/cli.test.ts @@ -162,28 +162,36 @@ describe("cli end-to-end", () => { }); // 2119: REQ-004.3.2, REQ-004.3.5 - it("init appends the AGENTS.md workflow section exactly once, mentioning the CI backstop", () => { + it("init appends the contracted AGENTS.md workflow section exactly once", () => { const root = mkdtempSync(join(tmpdir(), "2119-agents-")); writeFileSync(join(root, "AGENTS.md"), "# My project\n"); run(root, ["init"]); run(root, ["init"]); const body = readFileSync(join(root, "AGENTS.md"), "utf8"); + const workflow = body.match(/([\s\S]*?)/)?.[1]; + expect(workflow).toBeDefined(); + const normalizedWorkflow = workflow!.replace(/\s+/g, " "); expect(body.match(//g)).toHaveLength(1); expect(body.match(//g)).toHaveLength(1); expect(body).toContain("# My project"); - // The mandated workflow content: spec-first planning, test annotations, - // judgment reviews, and the check gate. - expect(body).toContain("write or update a spec in `specs/` first"); - expect(body).toContain("RFC 2119 keyword"); - const marker = ["21", "19"].join(""); // avoid a literal self-annotation - expect(body).toContain(`\`// ${marker}: REQ-001.2.3\``); - expect(body).toContain("fresh-context subagent"); - expect(body).toMatch(/npx rfc2119 check.*must exit 0/s); - expect(body).toContain("CI runs the same check"); - // 0.6 topics: draft-time spec critique + reviewer diversity (REQ-004.3.2). - expect(body).toContain("critique the draft\nrequirements"); - expect(body).toContain("review --audit"); - expect(body).toContain("different providers"); + expect(body.indexOf("")).toBeGreaterThan(body.indexOf("# My project")); + const instructions = [ + "write or update a spec in `specs/` first", + "observable product behavior and narrowly scoped text actually delivered", + "Keep test strategy, CI commands, review procedure, migration bookkeeping, and implementation notes non-normative", + "Consolidate duplicate requirements", + "dispatch a fresh-context reviewer to critique the draft", + "test annotated with a comment containing its ID", + "fresh-context reviewer judges each test's honesty", + "One behavioral test may cover multiple requirement IDs", + "add annotations or cross-references when the evidence is already sufficient, rather than duplicating the test", + "Reviewer diversity", + "run `npx rfc2119 check`", + "CI runs the same check", + ]; + for (const instruction of instructions) { + expect(normalizedWorkflow).toContain(instruction); + } }); // 2119: REQ-003.5.2, REQ-003.5.5 diff --git a/tests/rigor.test.ts b/tests/rigor.test.ts index 4009126..323c581 100644 --- a/tests/rigor.test.ts +++ b/tests/rigor.test.ts @@ -163,16 +163,29 @@ describe("deterministic rigor (0.6)", () => { expect(r.stderr).toContain("docs/autth/**"); }); - // 2119: REQ-003.1.10, REQ-003.1.11 - it("instruction files carry the counterexample obligation and bad-requirement clause", () => { + // 2119: REQ-003.1.10, REQ-003.1.11, REQ-003.1.12, REQ-003.1.13, REQ-003.1.14, REQ-003.1.15 + it("instruction files demand discriminating evidence without rewarding duplication", () => { const root = fixture(); run(root, ["review"]); const dir = join(root, ".2119/reviews"); const body = readFileSync(join(dir, readdirSync(dir)[0]), "utf8"); - expect(body).toContain("Counterexample obligation"); - expect(body).toMatch(/nearest violating\s+input/); - expect(body).toContain("not a pass"); - expect(body).toMatch(/bad\s+requirement honestly tested is still a bad requirement/); + const normalizedBody = body.replace(/\s+/g, " "); + const instructions = [ + /concrete implementation change that violates the requirement.*evidence would fail/s, + /quantifies over a set or names a defined grammar.*boundary members or edge productions/s, + /Do not demand a counterexample for every word or a Cartesian product of inputs/s, + /legitimate change that preserves the requirement's meaning.*evidence would stay green/s, + /One evidence body may cover multiple requirement IDs.*not a duplicate test/s, + /Reject evidence whose only value is pinning irrelevant wording, layout, digests, or implementation organization.*Preserve legitimate contracts.*inventories derived from the real product that fail loudly on zero subjects.*snapshots with an explicit, inexpensive update path/s, + /meaningfully distinct production behavior.*not a reason to demand every spelling or combination/s, + /ambiguous, untestable, or.*implementation mechanism rather than an observable outcome, fail with that finding/s, + ]; + for (const instruction of instructions) { + expect(normalizedBody).toMatch(instruction); + } + expect(normalizedBody).not.toMatch( + /(?:do not|never|prohibit(?:ed)?).{0,30}fail|fail.{0,30}(?:prohibit(?:ed)?|pass instead)/i, + ); }); // 2119: REQ-003.5.6 @@ -229,6 +242,7 @@ describe("deterministic rigor (0.6)", () => { expect(body).toContain("Adversarial Audit"); expect(body).toMatch(/concrete mutant or input/i); expect(body).toMatch(/violated while every\s+covering test stays green/); + expect(body).toMatch(/(?:scan|inspect|enumerate).*conjunct.*boundar.*precedence.*grammar.*(?:data )?shape/is); // The pass-only-if-no-counterexample directive is present. expect(body).toMatch(/Only if you genuinely cannot construct one/); expect(body.split("\n").filter((line) => /\bpass(?:ed)?\b/i.test(line))).toEqual([ @@ -236,49 +250,6 @@ describe("deterministic rigor (0.6)", () => { "- Only if you genuinely cannot construct one after honest effort: record a PASS stating the", `npx rfc2119 pass ${id} --summary "audit: "`, ]); - expect(body).toBe( - [ - "# 2119 Adversarial Audit: FIX-001.1.1", - "", - "This requirement's review previously PASSED. You are the adversary: your job is to break that", - "verdict, not to confirm it. You did not write the code or the tests under audit.", - "", - "## Requirement", - "", - "> The widget MUST spin.", - "", - "*(FIX-001.1.1, keyword: MUST)*", - "", - "## Evidence files", - "", - "- tests/widget.test.js", - "", - "## Your task", - "", - "**Construct a concrete mutant or input under which this requirement is violated while every", - "covering test stays green.** Enumerate the requirement's conjuncts and boundary terms; probe the", - "negative space (what must be refused, not what is accepted); consider shared fixtures, preludes,", - "and paths the tests never touch. Reason from the requirement's text, never from the", - "implementation's current behavior.", - "", - "- If you find such a counterexample: record a FAIL with the mutant described concretely enough", - " to reproduce.", - "- Only if you genuinely cannot construct one after honest effort: record a PASS stating the", - " strongest candidate you tried and why it fails to survive.", - "", - "## Recording your verdict", - "", - "Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim.", - "", - "```", - `npx rfc2119 pass ${id} --summary "audit: "`, - `npx rfc2119 fail ${id} --summary "audit: "`, - "```", - "", - "Do not edit any files; report, don't fix.", - "", - ].join("\n"), - ); expect(readFileSync(verdictPath, "utf8")).toBe(verdictBefore); expect(readFileSync(failingVerdictPath, "utf8")).toBe(failingVerdictBefore); expect(readFileSync(orphanVerdictPath, "utf8")).toBe(orphanVerdict); diff --git a/tests/self-supplied-evidence.test.ts b/tests/self-supplied-evidence.test.ts index 7db1db4..a96ba54 100644 --- a/tests/self-supplied-evidence.test.ts +++ b/tests/self-supplied-evidence.test.ts @@ -1,173 +1,13 @@ import { spawnSync } from "node:child_process"; -import { cpSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, writeFileSync } from "node:fs"; +import { cpSync, mkdirSync, mkdtempSync, realpathSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join, resolve } from "node:path"; -import { beforeAll, describe, expect, it } from "vitest"; -import { buildContext } from "../src/check.js"; +import { describe, expect, it } from "vitest"; const CLI = resolve(import.meta.dirname, "../dist/cli.js"); const REPO = resolve(import.meta.dirname, ".."); -const EXPECTED_PROVENANCE = `1. Name the concrete production failure this test would catch. - Cite file:line evidence that production can reach that failure without the test, fixtures, or prompts supplying the trigger or decisive observation. -2. Trace each applicable production boundary with file:line evidence. - A producer/consumer boundary means consuming a value emitted by a separately invoked production component or production data source. - If that boundary exists, cite file:line evidence that the test obtains its input from that producer. - If that boundary exists, cite file:line evidence that the exercised value preserves the producer's production shape. - If the decisive observation can equal an initial/default/placeholder/sentinel value, cite file:line evidence that the test distinguishes a newly produced observation from that pre-existing value. - A gate/runtime-environment boundary means invoking a binary or service outside the gate's own process. - If that boundary exists, cite file:line evidence for both its production provisioning declaration and the production path that fails when it is absent. -Record FAIL when applicable provenance evidence is absent or shows that production cannot produce the failure independently of the test setup.`; -const EXPECTED_TASK = `**Would the covering tests fail if this requirement were violated?** - -Read the requirement and each evidence file's tests annotated with \`2119: \` (or its section ID). Judge whether they genuinely verify the requirement. You MUST flag: - -- **Tautological assertions** — tests that assert what they just set up, or that cannot fail. -- **Over-mocking** — mocks/stubs that bypass the very behavior the requirement constrains. -- **Unrelated assertions** — tests that reference the requirement ID but assert something other than its criterion. -- **Keyword theater** — string/keyword matching standing in for behavioral verification. - -**Required production-provenance answers (a PASS is forbidden without them):** - -${EXPECTED_PROVENANCE} - -**Counterexample obligation:** enumerate the requirement's conjuncts and boundary terms (words -like "comment", "exactly", "only", "begins with"). For each, construct the nearest violating -input — the almost-conforming case the requirement forbids — and confirm a test rejects it. -When a requirement names a grammar or other defined input language, enumerate and probe its edge -productions rather than accepting coverage of only the most common form. -Do not reason from the implementation's current behavior; reason from the requirement's text. -A review that cannot name a rejected counterexample for a boundary term is not a pass. - -**Judge the requirement too:** if the requirement itself is ambiguous, untestable, or states an -implementation mechanism rather than an observable outcome, fail with that finding — a bad -requirement honestly tested is still a bad requirement. - -## Recording your verdict - -Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim. - -If the requirement's verification is genuine (or all findings were fixed), run: - -\`\`\` -npx rfc2119 pass --summary "" -\`\`\` - -If there are unresolved findings, run: - -\`\`\` -npx rfc2119 fail --summary "" -\`\`\` - -The summary is committed to the repository and read by humans in PR review — -be specific. Do not edit any files; report, don't fix.`; -const RECORDING_GUIDANCE = "Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim."; -const EXPECTED_DIRECT_TASK = `**Is this requirement genuinely satisfied by the current state of the evidence files?** - -Read the requirement and the evidence files and judge compliance directly. This requirement was tagged \`[review]\` because it needs judgment rather than a test. - -**Judge the requirement too:** if the requirement itself is ambiguous, untestable, or states an -implementation mechanism rather than an observable outcome, fail with that finding — a bad -requirement honestly tested is still a bad requirement. - -## Recording your verdict - -${RECORDING_GUIDANCE} - -If the requirement's verification is genuine (or all findings were fixed), run: - -\`\`\` -npx rfc2119 pass --summary "" -\`\`\` - -If there are unresolved findings, run: - -\`\`\` -npx rfc2119 fail --summary "" -\`\`\` - -The summary is committed to the repository and read by humans in PR review — -be specific. Do not edit any files; report, don't fix.`; -const EXPECTED_AUDIT_TASK = `**Construct a concrete mutant or input under which this requirement is violated while every -covering test stays green.** Enumerate the requirement's conjuncts and boundary terms; probe the -negative space (what must be refused, not what is accepted); consider shared fixtures, preludes, -and paths the tests never touch. Reason from the requirement's text, never from the -implementation's current behavior. - -- If you find such a counterexample: record a FAIL with the mutant described concretely enough - to reproduce. -- Only if you genuinely cannot construct one after honest effort: record a PASS stating the - strongest candidate you tried and why it fails to survive. - -## Recording your verdict - -${RECORDING_GUIDANCE} - -\`\`\` -npx rfc2119 pass --summary "audit: " -npx rfc2119 fail --summary "audit: " -\`\`\` - -Do not edit any files; report, don't fix.`; - -function normalizedTask(body: string): string { - return body - .replace(/\n## Additional review criteria\n[\s\S]*?(?=\n## Recording your verdict)/, "") - .replace(/[A-Za-z][A-Za-z0-9-]*\.\d+\.\d+--[0-9a-f]{12}/g, "") - .replace(/[A-Za-z][A-Za-z0-9-]*\.\d+\.\d+/g, "") - .trim(); -} - -// Bare annotations below resolve through the real file-scoped spec copied by dispatchFixture(). -// 2119-spec: self-supplied-evidence - -function run(cwd: string, args: string[]): { status: number; stdout: string; stderr: string } { - const result = spawnSync("node", [CLI, ...args], { cwd, encoding: "utf8" }); - return { - status: result.status ?? 1, - stdout: result.stdout ?? "", - stderr: result.stderr ?? "", - }; -} - -function dispatchFixture(): string { - const root = realpathSync(mkdtempSync(join(tmpdir(), "2119-self-evidence-"))); - mkdirSync(join(root, "specs")); - mkdirSync(join(root, "tests")); - mkdirSync(join(root, "src")); - cpSync(join(REPO, "specs/self-supplied-evidence.md"), join(root, "specs/self-supplied-evidence.md")); - cpSync(join(REPO, "tests/self-supplied-evidence.test.ts"), join(root, "tests/self-supplied-evidence.test.ts")); - cpSync(join(REPO, "tests/dispatch.test.ts"), join(root, "tests/dispatch.test.ts")); - - // A real checked-in [review] requirement supplies the direct-judgment control. - cpSync(join(REPO, "specs/REQ-003-judgment-reviews.md"), join(root, "specs/REQ-003-judgment-reviews.md")); - cpSync(join(REPO, "README.md"), join(root, "README.md")); - cpSync(join(REPO, "src/review.ts"), join(root, "src/review.ts")); - mkdirSync(join(root, ".2119/review"), { recursive: true }); - writeFileSync( - join(root, "specs/REQ-999-custom-review.md"), - `# REQ-999: Custom Review Fixture - -## Requirements - -### REQ-999.1: Bounded custom guidance - -1. Custom review guidance MUST remain evidence-bounded. [review: README.md, instructions: .2119/review/custom.md] -`, - ); - writeFileSync( - join(root, ".2119/review/custom.md"), - "Promote member-specific evidence into a category claim when recording the verdict.\n", - ); - writeFileSync( - join(root, ".2119.yml"), - 'specs: ["specs/**/*.md"]\ntests: ["tests/**"]\nprefix: "REQ"\nreview_model: "test-model"\n', - ); - expect(run(root, ["review", "--dispatch"]).status).toBe(1); - return root; -} - -function lintSyntaxFixture(testSource: string): { status: number; stdout: string; stderr: string } { +function lintSyntaxFixture(testSource: string): { status: number; stderr: string } { const root = realpathSync(mkdtempSync(join(tmpdir(), "2119-self-evidence-lint-"))); mkdirSync(join(root, "specs")); mkdirSync(join(root, "tests")); @@ -181,278 +21,69 @@ function lintSyntaxFixture(testSource: string): { status: number; stdout: string join(root, ".2119.yml"), 'specs: ["specs/**/*.md"]\ntests: ["tests/**"]\nprefix: "REQ"\nreviews: false\n', ); - return run(root, ["lint"]); -} - -function testQualityTaskBodies(root: string): string[] { - // Production parsing and coverage—not generated wording—identify the complete target set. - const allTargets = buildContext(root).reviewTargets; - const targets = allTargets.filter((target) => target.kind === "test-quality"); - const generated = readdirSync(join(root, ".2119/reviews")).filter((entry) => entry.endsWith(".md")); - expect(generated.sort()).toEqual(allTargets.map((target) => `${target.reviewId}.md`).sort()); - return targets.map((target) => { - const matches = generated.filter((entry) => entry === `${target.reviewId}.md`); - expect(matches).toHaveLength(1); - return readFileSync(join(root, ".2119/reviews", matches[0]), "utf8").split("## Your task\n\n", 2)[1]; - }); -} - -function expectEveryTestQualityTask(root: string, assertion: (body: string) => void): void { - const bodies = testQualityTaskBodies(root); - expect(bodies.length).toBeGreaterThan(2); - for (const body of bodies) { - expect(normalizedTask(body)).toBe(EXPECTED_TASK); - const provenance = body - .split("**Required production-provenance answers (a PASS is forbidden without them):**", 2)[1] - ?.split("**Counterexample obligation:**", 1)[0]; - expect(provenance?.trim()).toBe(EXPECTED_PROVENANCE); - assertion(body); - } + const result = spawnSync("node", [CLI, "lint"], { cwd: root, encoding: "utf8" }); + return { status: result.status ?? 1, stderr: result.stderr ?? "" }; } -describe("self-supplied evidence review instructions", () => { - let root: string; - - beforeAll(() => { - root = dispatchFixture(); - }); - - // 2119: 1.1 - it("asks for the concrete production failure", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toMatch(/^1\. Name the concrete production failure this test would catch\.$/m); - }); - }); - - // 2119: 1.2 - it("demands a file:line production reachability trace independent of test setup", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toMatch( - /^ Cite file:line evidence that production can reach that failure without the test, fixtures, or prompts supplying the trigger or decisive observation\.$/m, - ); - }); - }); - - // 2119: 2.1 - it("defines the producer/consumer boundary narrowly", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toMatch( - /^ A producer\/consumer boundary means consuming a value emitted by a separately invoked production component or production data source\.$/m, - ); - }); - }); - - // 2119: 2.2 - it("requires applicable tests to source input from the production producer", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toContain("cite file:line evidence that the test obtains its input from that producer"); - }); - }); - - // 2119: 2.3 - it("requires applicable tests to preserve the producer's value shape", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toContain("cite file:line evidence that the exercised value preserves the producer's production shape"); - }); - }); - - // 2119: 2.4 - it("requires a new observation to be distinguished from pre-existing sentinels", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toMatch( - /^ If the decisive observation can equal an initial\/default\/placeholder\/sentinel value, cite file:line evidence that the test distinguishes a newly produced observation from that pre-existing value\.$/m, - ); - }); - }); - - // 2119: 3.1 - it("defines the runtime-environment boundary narrowly", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toMatch( - /^ A gate\/runtime-environment boundary means invoking a binary or service outside the gate's own process\.$/m, - ); - }); - }); - - // 2119: 3.2 - it("requires provisioning and absence-path evidence for external dependencies", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toMatch( - /^ If that boundary exists, cite file:line evidence for both its production provisioning declaration and the production path that fails when it is absent\.$/m, - ); - }); - }); - - // 2119: 4.1 - it("makes missing or self-supplied provenance a failing verdict", () => { - expectEveryTestQualityTask(root, (body) => { - expect(body).toContain("Record FAIL when applicable provenance evidence is absent or shows that production cannot produce the failure independently of the test setup."); - }); - }); - - // 2119: 5.1 - it("keeps the test-quality provenance questions out of direct judgments", () => { - const targets = buildContext(root).reviewTargets.filter((target) => target.kind === "requirement"); - expect(targets.length).toBeGreaterThan(0); - for (const target of targets) { - const direct = readFileSync(join(root, ".2119/reviews", `${target.reviewId}.md`), "utf8"); - expect(normalizedTask(direct.split("## Your task\n\n", 2)[1])).toBe(EXPECTED_DIRECT_TASK); - expect(direct).not.toContain("Required production-provenance answers"); - } - }); - +describe("self-supplied evidence lint compatibility", () => { + // 2119-spec: self-supplied-evidence // 2119: 6.1 - it("does not infer provenance lint failures from literal or factory syntax", () => { - const syntaxVariants = [ - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses scalar literal input", () => { + it("accepts representative literal and factory input syntax", () => { + const testBodies = [ + `it("uses scalar literal input", () => { const input = "literal input"; expect(input).toBe("literal input"); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses object literal input", () => { +});`, + `it("uses object literal input", () => { const input = { source: "literal input" }; expect(input.source).toBe("literal input"); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses array literal input", () => { +});`, + `it("uses array literal input", () => { const input = ["literal input"]; expect(input[0]).toBe("literal input"); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses template literal input", () => { +});`, + `it("uses template literal input", () => { const input = \`template input\`; expect(input).toBe("template input"); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses regexp literal input", () => { +});`, + `it("uses regexp literal input", () => { const input = /literal input/; expect(input.test("literal input")).toBe(true); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses numeric literal input", () => { - const input = 42; - expect(input).toBe(42); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses bigint literal input", () => { - const input = 42n; - expect(input).toBe(42n); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses boolean literal input", () => { - const input = true; - expect(input).toBe(true); -}); -`, - `// 2119-spec: self-supplied-evidence -// 2119: 6.1 -it("uses null literal input", () => { - const input = null; - expect(input).toBeNull(); -}); -`, - `// 2119-spec: self-supplied-evidence -function makeInput(value: string): string { return String(value); } -// 2119: 6.1 +});`, + `it("uses primitive literal input", () => { + expect([42, 42n, true, null]).toHaveLength(4); +});`, + `function makeInput(value: string): string { return String(value); } it("uses named factory input", () => { - const input = makeInput("factory input"); - expect(input).toBe("factory input"); -}); -`, - `// 2119-spec: self-supplied-evidence -class InputFactory { static create(value: string): string { return String(value); } } -// 2119: 6.1 -it("uses class factory input", () => { - const input = InputFactory.create("factory input"); - expect(input).toBe("factory input"); -}); -`, - `// 2119-spec: self-supplied-evidence -import { importedFactory } from "./factory.js"; -// 2119: 6.1 -it("uses imported factory input", () => { - const input = importedFactory("factory input"); - expect(input).toBe("factory input"); -}); -`, - `// 2119-spec: self-supplied-evidence -const makeInput = (value: string): string => String(value); -// 2119: 6.1 + expect(makeInput("factory input")).toBe("factory input"); +});`, + `class StaticFactory { static create(value: string): string { return String(value); } } +it("uses static factory input", () => { + expect(StaticFactory.create("factory input")).toBe("factory input"); +});`, + `class ConstructorFactory { constructor(readonly value: string) {} } +it("uses constructor factory input", () => { + expect(new ConstructorFactory("factory input").value).toBe("factory input"); +});`, + `class MethodFactory { build(value: string): string { return String(value); } } +it("uses instance-method factory input", () => { + expect(new MethodFactory().build("factory input")).toBe("factory input"); +});`, + `it("uses imported factory input", () => { + expect(importedFactory("factory input")).toBe("factory input"); +});`, + `const arrowFactory = (value: string): string => String(value); it("uses arrow factory input", () => { - const input = makeInput("factory input"); - expect(input).toBe("factory input"); -}); -`, + expect(arrowFactory("factory input")).toBe("factory input"); +});`, ]; - for (const testSource of syntaxVariants) { - const result = lintSyntaxFixture(testSource); - expect(result.status).toBe(0); - expect(result.stdout).toBe("lint: 1 spec file(s) clean\n"); - expect(result.stderr).toBe(""); - } - }); - - // 2119: 7.1 - it("bounds every standard fixture target and every audit generated from committed verdicts", () => { - const expectBoundedGuidance = (body: string): void => { - const recordingGuidance = body.split("## Recording your verdict\n\n", 2)[1]; - expect(recordingGuidance).toMatch( - /^Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular\/plural scope; do not promote member-specific evidence into a category claim\.$/m, - ); - expect(recordingGuidance).not.toMatch( - /ignore|disregard|optional|need not|not required|may (?:broaden|generalize|promote)|broader (?:scope|category) (?:is|remains) allowed/i, - ); - }; - - const targets = buildContext(root).reviewTargets; - expect(targets.length).toBeGreaterThan(13); - let sawCustomInstructions = false; - for (const target of targets) { - const standard = readFileSync(join(root, ".2119/reviews", `${target.reviewId}.md`), "utf8"); - expectBoundedGuidance(standard); - if (standard.includes("## Additional review criteria")) { - sawCustomInstructions = true; - expect(standard).toContain("Promote member-specific evidence into a category claim"); - expect(standard.indexOf("## Your task")).toBeGreaterThan( - standard.indexOf("## Additional review criteria"), - ); - } - expect(normalizedTask(standard.split("## Your task\n\n", 2)[1])).toBe( - target.kind === "test-quality" ? EXPECTED_TASK : EXPECTED_DIRECT_TASK, - ); - } - expect(sawCustomInstructions).toBe(true); - - cpSync(join(REPO, ".2119/verdicts"), join(root, ".2119/verdicts"), { recursive: true }); - const withVerdicts = buildContext(root); - const passingTargets = withVerdicts.reviewTargets.filter( - (target) => withVerdicts.verdicts.get(target.reviewId)?.verdict === "pass", - ); - expect(passingTargets.length).toBeGreaterThan(2); - expect(run(root, ["review", "--audit", "--dispatch"]).status).toBe(1); + const source = `// 2119-spec: self-supplied-evidence +import { importedFactory } from "./factory.js"; - const auditNames = readdirSync(join(root, ".2119/reviews")).filter((name) => name.endsWith(".audit.md")); - expect(auditNames.sort()).toEqual(passingTargets.map((target) => `${target.reviewId}.audit.md`).sort()); - for (const auditName of auditNames) { - const audit = readFileSync(join(root, ".2119/reviews", auditName), "utf8"); - expectBoundedGuidance(audit); - expect(normalizedTask(audit.split("## Your task\n\n", 2)[1])).toBe(EXPECTED_AUDIT_TASK); - } +${testBodies.map((body) => `// 2119: 6.1\n${body}`).join("\n\n")} +`; + const result = lintSyntaxFixture(source); + expect(result.status).toBe(0); + expect(result.stderr).toBe(""); }); });