diff --git a/.2119/verdicts/REQ-003.2.2--bea345f6b733.json b/.2119/verdicts/REQ-003.2.2--bea345f6b733.json new file mode 100644 index 0000000..4e7f744 --- /dev/null +++ b/.2119/verdicts/REQ-003.2.2--bea345f6b733.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.2.2--bea345f6b733", + "requirementId": "REQ-003.2.2", + "hash": "bea345f6b733", + "verdict": "pass", + "summary": "pass and fail serialize verdict records as plain JSON in .2119/verdicts/, while init and verdict writes repair ignore rules so those records remain trackable.", + "timestamp": "2026-08-02T22:29:35.122Z" +} diff --git a/.2119/verdicts/REQ-003.8.1--683e7215f03e.json b/.2119/verdicts/REQ-003.8.1--683e7215f03e.json new file mode 100644 index 0000000..53a3a4a --- /dev/null +++ b/.2119/verdicts/REQ-003.8.1--683e7215f03e.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.1--683e7215f03e", + "requirementId": "REQ-003.8.1", + "hash": "683e7215f03e", + "verdict": "fail", + "summary": "eval/calibration's 13 cases omit the repository's documented task_lib.sh-to-'shell libraries' verdict-scope inflation escape, so the corpus no longer contains every known review escape.", + "timestamp": "2026-08-02T22:32:53.245Z" +} diff --git a/.2119/verdicts/REQ-003.8.1--87a0d11bd0c2.json b/.2119/verdicts/REQ-003.8.1--87a0d11bd0c2.json new file mode 100644 index 0000000..0b34e97 --- /dev/null +++ b/.2119/verdicts/REQ-003.8.1--87a0d11bd0c2.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.1--87a0d11bd0c2", + "requirementId": "REQ-003.8.1", + "hash": "87a0d11bd0c2", + "verdict": "pass", + "summary": "The corpus fixtures retain the required fields and documented escapes, while corrected case 013 now accurately condenses missing/invalid verdict rejection and real-CLI acceptance from its cited production test.", + "timestamp": "2026-08-02T23:02:34.369Z" +} diff --git a/.2119/verdicts/REQ-003.8.1--b858103eb35c.json b/.2119/verdicts/REQ-003.8.1--b858103eb35c.json new file mode 100644 index 0000000..f33b708 --- /dev/null +++ b/.2119/verdicts/REQ-003.8.1--b858103eb35c.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.1--b858103eb35c", + "requirementId": "REQ-003.8.1", + "hash": "b858103eb35c", + "verdict": "fail", + "summary": "The listed fixtures have the required fields, but 'every known review escape' has no defined external inventory or scope, so corpus completeness is ambiguous and cannot be verified from the corpus's self-claim.", + "timestamp": "2026-08-02T22:31:09.428Z" +} diff --git a/.2119/verdicts/REQ-003.8.1--c355ec3bc765.json b/.2119/verdicts/REQ-003.8.1--c355ec3bc765.json new file mode 100644 index 0000000..c58f040 --- /dev/null +++ b/.2119/verdicts/REQ-003.8.1--c355ec3bc765.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.1--c355ec3bc765", + "requirementId": "REQ-003.8.1", + "hash": "c355ec3bc765", + "verdict": "pass", + "summary": "The calibration corpus contains the required fixture fields and now includes 014 for the documented PR #112 singular-evidence-to-plural-category scope-inflation escape.", + "timestamp": "2026-08-02T22:34:46.490Z" +} diff --git a/.2119/verdicts/REQ-003.8.2--5262e7c71e7b.json b/.2119/verdicts/REQ-003.8.2--5262e7c71e7b.json new file mode 100644 index 0000000..98d9ee8 --- /dev/null +++ b/.2119/verdicts/REQ-003.8.2--5262e7c71e7b.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.2--5262e7c71e7b", + "requirementId": "REQ-003.8.2", + "hash": "5262e7c71e7b", + "verdict": "fail", + "summary": "The current template cannot preserve calibration case 013's expected PASS: its conjunct rule requires rejected counterexamples for empty note, mismatched owner, and unparseable stamp, which that case's evidence does not provide.", + "timestamp": "2026-08-02T22:33:18.373Z" +} diff --git a/.2119/verdicts/REQ-003.8.2--5ec997925e8d.json b/.2119/verdicts/REQ-003.8.2--5ec997925e8d.json new file mode 100644 index 0000000..09c2ee1 --- /dev/null +++ b/.2119/verdicts/REQ-003.8.2--5ec997925e8d.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.2--5ec997925e8d", + "requirementId": "REQ-003.8.2", + "hash": "5ec997925e8d", + "verdict": "fail", + "summary": "Calibration case 013-pass-control-negative-positive expects PASS, but the template requires a rejected counterexample for every conjunct while its evidence never independently rejects an empty note, mismatched owner, or unparseable stamp; a reviewer following the template must FAIL that case.", + "timestamp": "2026-08-02T22:31:16.326Z" +} diff --git a/.2119/verdicts/REQ-003.8.2--d52ce5a3aa84.json b/.2119/verdicts/REQ-003.8.2--d52ce5a3aa84.json new file mode 100644 index 0000000..d0137cf --- /dev/null +++ b/.2119/verdicts/REQ-003.8.2--d52ce5a3aa84.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.2--d52ce5a3aa84", + "requirementId": "REQ-003.8.2", + "hash": "d52ce5a3aa84", + "verdict": "pass", + "summary": "The current standard and audit instructions retain the checks needed by calibration cases 001-014: tautology/over-mocking flags, provenance, conjunct and boundary counterexamples, requirement-quality judgment, and evidence-bounded summary scope; case 013 now rejects each named malformed-record conjunct and keeps a genuine-writer pass control.", + "timestamp": "2026-08-02T22:34:56.716Z" +} diff --git a/.2119/verdicts/REQ-003.8.2--deb61e2892be.json b/.2119/verdicts/REQ-003.8.2--deb61e2892be.json new file mode 100644 index 0000000..0308d2a --- /dev/null +++ b/.2119/verdicts/REQ-003.8.2--deb61e2892be.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.2--deb61e2892be", + "requirementId": "REQ-003.8.2", + "hash": "deb61e2892be", + "verdict": "fail", + "summary": "The test-quality template forbids PASS without production file:line provenance, but calibration cases 012-pass-control-perturbation and 013-pass-control-negative-positive contain only condensed snippets and expect PASS, so a reviewer following the template cannot preserve those two corpus verdicts.", + "timestamp": "2026-08-02T22:29:50.495Z" +} diff --git a/.2119/verdicts/REQ-003.8.2--ec2046ab17a0.json b/.2119/verdicts/REQ-003.8.2--ec2046ab17a0.json new file mode 100644 index 0000000..6f90f0b --- /dev/null +++ b/.2119/verdicts/REQ-003.8.2--ec2046ab17a0.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-003.8.2--ec2046ab17a0", + "requirementId": "REQ-003.8.2", + "hash": "ec2046ab17a0", + "verdict": "pass", + "summary": "The current standard, direct-judgment, and audit instructions preserve the expected verdicts for calibration cases 001-014; corrected case 013 now independently covers its missing verdict, invalid verdict, and real-CLI acceptance clauses.", + "timestamp": "2026-08-02T23:02:58.593Z" +} diff --git a/.2119/verdicts/REQ-004.1.3--f46aea2e9d6b.json b/.2119/verdicts/REQ-004.1.3--f46aea2e9d6b.json new file mode 100644 index 0000000..a29ea3d --- /dev/null +++ b/.2119/verdicts/REQ-004.1.3--f46aea2e9d6b.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.1.3--f46aea2e9d6b", + "requirementId": "REQ-004.1.3", + "hash": "f46aea2e9d6b", + "verdict": "fail", + "summary": "The tests handcraft hook payloads instead of obtaining a production platform payload, omit Codex Delete File apply_patch paths and custom/test-glob positive boundaries, and direct helper calls do not prove path determination occurs inside the CLI.", + "timestamp": "2026-08-02T22:31:15.380Z" +} diff --git a/.2119/verdicts/REQ-004.1.7--0cff14d8a0f3.json b/.2119/verdicts/REQ-004.1.7--0cff14d8a0f3.json new file mode 100644 index 0000000..26e441b --- /dev/null +++ b/.2119/verdicts/REQ-004.1.7--0cff14d8a0f3.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.1.7--0cff14d8a0f3", + "requirementId": "REQ-004.1.7", + "hash": "0cff14d8a0f3", + "verdict": "fail", + "summary": "The annotated test calls handleHook directly and never exercises generated platform hook wiring; deleting the SessionStart entries in src/adapters.ts would prevent session-start injection through the platform mechanism while every REQ-004.1.7 assertion stays green.", + "timestamp": "2026-08-02T22:33:04.234Z" +} diff --git a/.2119/verdicts/REQ-004.1.7--69cfdfd5ebdd.json b/.2119/verdicts/REQ-004.1.7--69cfdfd5ebdd.json new file mode 100644 index 0000000..9e9820e --- /dev/null +++ b/.2119/verdicts/REQ-004.1.7--69cfdfd5ebdd.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.1.7--69cfdfd5ebdd", + "requirementId": "REQ-004.1.7", + "hash": "69cfdfd5ebdd", + "verdict": "fail", + "summary": "The test does not execute the installed SessionStart command: a broken command can retain the asserted hook substring while the separately invoked dist/cli.js still returns context, leaving the platform integration broken.", + "timestamp": "2026-08-02T22:36:05.854Z" +} diff --git a/.2119/verdicts/REQ-004.1.7--9247f44ae9d9.json b/.2119/verdicts/REQ-004.1.7--9247f44ae9d9.json new file mode 100644 index 0000000..d7e462a --- /dev/null +++ b/.2119/verdicts/REQ-004.1.7--9247f44ae9d9.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.1.7--9247f44ae9d9", + "requirementId": "REQ-004.1.7", + "hash": "9247f44ae9d9", + "verdict": "fail", + "summary": "The test separately inspects installed SessionStart command text and calls handleHook, but never executes that command through the CLI; breaking src/cli.ts hook dispatch would prevent all three installed commands from injecting context while every annotated assertion stays green.", + "timestamp": "2026-08-02T22:34:39.173Z" +} diff --git a/.2119/verdicts/REQ-004.1.7--a4801c86dd0e.json b/.2119/verdicts/REQ-004.1.7--a4801c86dd0e.json new file mode 100644 index 0000000..a3874f3 --- /dev/null +++ b/.2119/verdicts/REQ-004.1.7--a4801c86dd0e.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.1.7--a4801c86dd0e", + "requirementId": "REQ-004.1.7", + "hash": "a4801c86dd0e", + "verdict": "fail", + "summary": "The test covers only Claude and derives the expected body from production SESSION_CONTEXT; Codex/Gemini omission or a non-descriptive/overlong context can remain green, while 'short' has no testable bound.", + "timestamp": "2026-08-02T22:29:38.198Z" +} diff --git a/.2119/verdicts/REQ-004.1.7--a6697c983bee.json b/.2119/verdicts/REQ-004.1.7--a6697c983bee.json new file mode 100644 index 0000000..47d4a8d --- /dev/null +++ b/.2119/verdicts/REQ-004.1.7--a6697c983bee.json @@ -0,0 +1,8 @@ +{ + "reviewId": "REQ-004.1.7--a6697c983bee", + "requirementId": "REQ-004.1.7", + "hash": "a6697c983bee", + "verdict": "pass", + "summary": "For each installed Claude, Codex, and Gemini SessionStart command, the test executes that production command and requires successful JSON with SessionStart additionalContext naming spec-driven testing, MUST-level obligations, and the check command in under 1000 characters.", + "timestamp": "2026-08-02T22:37:23.963Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.1--3a42292f6705.json b/.2119/verdicts/self-supplied-evidence.1.1--3a42292f6705.json new file mode 100644 index 0000000..a6c955c --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.1--3a42292f6705.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.1--3a42292f6705", + "requirementId": "self-supplied-evidence.1.1", + "hash": "3a42292f6705", + "verdict": "pass", + "summary": "Every production-computed test-quality task is compared against the complete canonical task body, so removing, weakening, or contradicting the concrete-production-failure mandate fails the test.", + "timestamp": "2026-08-02T23:10:57.874Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.1--6419602f2307.json b/.2119/verdicts/self-supplied-evidence.1.1--6419602f2307.json new file mode 100644 index 0000000..002bebb --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.1--6419602f2307.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.1--6419602f2307", + "requirementId": "self-supplied-evidence.1.1", + "hash": "6419602f2307", + "verdict": "fail", + "summary": "The exact mandate is asserted, but the anti-waiver check misses contradictory guidance such as 'A generic category-level failure is sufficient,' which permits a non-concrete answer while the test stays green.", + "timestamp": "2026-08-02T23:07:14.248Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.1--7c4774226fe5.json b/.2119/verdicts/self-supplied-evidence.1.1--7c4774226fe5.json new file mode 100644 index 0000000..bcb0d2a --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.1--7c4774226fe5.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.1--7c4774226fe5", + "requirementId": "self-supplied-evidence.1.1", + "hash": "7c4774226fe5", + "verdict": "pass", + "summary": "Every production-identified test-quality instruction is required to ask for a concrete production failure, while tests reject omission, optionality, and contradictory carve-outs.", + "timestamp": "2026-08-02T05:32:14.238Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.1--8e33566ca8c8.json b/.2119/verdicts/self-supplied-evidence.1.1--8e33566ca8c8.json new file mode 100644 index 0000000..c0a1992 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.1--8e33566ca8c8.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.1--8e33566ca8c8", + "requirementId": "self-supplied-evidence.1.1", + "hash": "8e33566ca8c8", + "verdict": "fail", + "summary": "The provenance slice is exact, but contradictory guidance appended after the counterexample block can permit a generic failure category while every 1.1 assertion remains green.", + "timestamp": "2026-08-02T23:08:33.884Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.1--ef8087c8df76.json b/.2119/verdicts/self-supplied-evidence.1.1--ef8087c8df76.json new file mode 100644 index 0000000..f018075 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.1--ef8087c8df76.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.1--ef8087c8df76", + "requirementId": "self-supplied-evidence.1.1", + "hash": "ef8087c8df76", + "verdict": "pass", + "summary": "Every production-computed test-quality instruction retains the complete canonical task body, so removing, weakening, or contradicting the concrete-production-failure mandate fails the test.", + "timestamp": "2026-08-02T23:14:00.257Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.2--419abd7dee19.json b/.2119/verdicts/self-supplied-evidence.1.2--419abd7dee19.json new file mode 100644 index 0000000..c0c686a --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.2--419abd7dee19.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.2--419abd7dee19", + "requirementId": "self-supplied-evidence.1.2", + "hash": "419abd7dee19", + "verdict": "pass", + "summary": "Every production-discovered test-quality instruction is compared against the complete expected task body with only dynamic IDs normalized, so removing, weakening, or contradicting the file:line independence demand fails.", + "timestamp": "2026-08-02T23:10:46.134Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.2--4f90050b0b09.json b/.2119/verdicts/self-supplied-evidence.1.2--4f90050b0b09.json new file mode 100644 index 0000000..66d9520 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.2--4f90050b0b09.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.2--4f90050b0b09", + "requirementId": "self-supplied-evidence.1.2", + "hash": "4f90050b0b09", + "verdict": "pass", + "summary": "Every production-discovered test-quality instruction is still matched against the complete expected task with only dynamic IDs normalized, so any omission, weakening, or contradiction of the file:line production-reachability demand fails.", + "timestamp": "2026-08-02T23:14:09.042Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.2--6331058954ea.json b/.2119/verdicts/self-supplied-evidence.1.2--6331058954ea.json new file mode 100644 index 0000000..7535f16 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.2--6331058954ea.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.2--6331058954ea", + "requirementId": "self-supplied-evidence.1.2", + "hash": "6331058954ea", + "verdict": "pass", + "summary": "All generated test-quality tasks are exhaustively selected from production review targets and require file:line production reachability for the named failure, with exact independence from test, fixture, and prompt-supplied triggers or decisive observations and explicit rejection of opt-outs.", + "timestamp": "2026-08-02T05:32:24.495Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.2--78804c69a5b9.json b/.2119/verdicts/self-supplied-evidence.1.2--78804c69a5b9.json new file mode 100644 index 0000000..5df7911 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.2--78804c69a5b9.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.2--78804c69a5b9", + "requirementId": "self-supplied-evidence.1.2", + "hash": "78804c69a5b9", + "verdict": "fail", + "summary": "The test stays green if the provenance section retains the exact mandate but adds 'Evidence from test fixtures satisfies this demand.'; that unrecognized contradiction permits fixture-supplied evidence despite 1.2's independence clause.", + "timestamp": "2026-08-02T23:07:12.165Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.1.2--82dcac5dc17c.json b/.2119/verdicts/self-supplied-evidence.1.2--82dcac5dc17c.json new file mode 100644 index 0000000..dda4be0 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.1.2--82dcac5dc17c.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.1.2--82dcac5dc17c", + "requirementId": "self-supplied-evidence.1.2", + "hash": "82dcac5dc17c", + "verdict": "fail", + "summary": "The exact comparison ends at '**Counterexample obligation:**'; appending later in the same instruction 'Evidence from test fixtures satisfies the production-provenance demand.' leaves every assertion green while negating 1.2's independence requirement.", + "timestamp": "2026-08-02T23:08:31.147Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.1--1fc1ddf5c224.json b/.2119/verdicts/self-supplied-evidence.2.1--1fc1ddf5c224.json new file mode 100644 index 0000000..94d303c --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.1--1fc1ddf5c224.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.1--1fc1ddf5c224", + "requirementId": "self-supplied-evidence.2.1", + "hash": "1fc1ddf5c224", + "verdict": "pass", + "summary": "The real review-dispatch CLI generates every computed test-quality instruction, and the test rejects omission or alteration of consumption, emitted value, separate invocation, production component, or production data-source terms as well as weakening exceptions.", + "timestamp": "2026-08-02T23:07:04.908Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.1--2aa103ac58bd.json b/.2119/verdicts/self-supplied-evidence.2.1--2aa103ac58bd.json new file mode 100644 index 0000000..6b2b59f --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.1--2aa103ac58bd.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.1--2aa103ac58bd", + "requirementId": "self-supplied-evidence.2.1", + "hash": "2aa103ac58bd", + "verdict": "pass", + "summary": "All generated test-quality tasks are enumerated from production review targets and must contain the exact narrow boundary definition; omission or alteration of consumption, emitted value, separate invocation, or production component/data-source terms fails, while anti-waiver assertions reject contradictory weakening.", + "timestamp": "2026-08-02T05:32:29.175Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.1--611ec8687d7d.json b/.2119/verdicts/self-supplied-evidence.2.1--611ec8687d7d.json new file mode 100644 index 0000000..796d3bd --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.1--611ec8687d7d.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.1--611ec8687d7d", + "requirementId": "self-supplied-evidence.2.1", + "hash": "611ec8687d7d", + "verdict": "pass", + "summary": "Every computed production-generated test-quality task is compared with a complete oracle after only dynamic ID normalization, so changing consumption, emitted value, separate invocation, production-component, or production-data-source scope fails.", + "timestamp": "2026-08-02T23:14:10.359Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.1--7da77538aed0.json b/.2119/verdicts/self-supplied-evidence.2.1--7da77538aed0.json new file mode 100644 index 0000000..fc9bc91 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.1--7da77538aed0.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.1--7da77538aed0", + "requirementId": "self-supplied-evidence.2.1", + "hash": "7da77538aed0", + "verdict": "pass", + "summary": "Every production-dispatched test-quality instruction must match an independent full provenance oracle, so altering any producer/consumer boundary conjunct—consumption, emitted value, separate invocation, production component, or production data source—fails.", + "timestamp": "2026-08-02T23:08:26.294Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.1--a000c7a0aa4b.json b/.2119/verdicts/self-supplied-evidence.2.1--a000c7a0aa4b.json new file mode 100644 index 0000000..38751c4 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.1--a000c7a0aa4b.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.1--a000c7a0aa4b", + "requirementId": "self-supplied-evidence.2.1", + "hash": "a000c7a0aa4b", + "verdict": "pass", + "summary": "Every generated test-quality task is matched against a complete independent task oracle after normalizing only requirement/review IDs, so any change to the producer/consumer definition's consumption, emitted value, separate invocation, production component, or data-source scope fails.", + "timestamp": "2026-08-02T23:10:44.573Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.2--156362cfdb04.json b/.2119/verdicts/self-supplied-evidence.2.2--156362cfdb04.json new file mode 100644 index 0000000..429ed15 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.2--156362cfdb04.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.2--156362cfdb04", + "requirementId": "self-supplied-evidence.2.2", + "hash": "156362cfdb04", + "verdict": "fail", + "summary": "The assertion checks only the citation clause substring: changing its trigger to 'Unless that boundary exists' leaves the test and anti-waiver regexes green while no longer requiring evidence whenever a producer/consumer boundary exists.", + "timestamp": "2026-08-02T23:07:37.188Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.2--56dbe92c6003.json b/.2119/verdicts/self-supplied-evidence.2.2--56dbe92c6003.json new file mode 100644 index 0000000..744e1b8 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.2--56dbe92c6003.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.2--56dbe92c6003", + "requirementId": "self-supplied-evidence.2.2", + "hash": "56dbe92c6003", + "verdict": "pass", + "summary": "Every production-computed test-quality task pins the producer/consumer condition and requires file:line evidence that the covering test obtains its input from that production producer.", + "timestamp": "2026-08-02T23:14:23.989Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.2--8fa7baeaefb2.json b/.2119/verdicts/self-supplied-evidence.2.2--8fa7baeaefb2.json new file mode 100644 index 0000000..406bb51 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.2--8fa7baeaefb2.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.2--8fa7baeaefb2", + "requirementId": "self-supplied-evidence.2.2", + "hash": "8fa7baeaefb2", + "verdict": "pass", + "summary": "The test inspects every generated test-quality task, requires mandatory file:line evidence that its input comes from the production producer, and rejects waiver or self-supplied-evidence language.", + "timestamp": "2026-08-02T05:33:28.165Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.2--a3e153ca68e4.json b/.2119/verdicts/self-supplied-evidence.2.2--a3e153ca68e4.json new file mode 100644 index 0000000..292ae94 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.2--a3e153ca68e4.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.2--a3e153ca68e4", + "requirementId": "self-supplied-evidence.2.2", + "hash": "a3e153ca68e4", + "verdict": "pass", + "summary": "Each generated test-quality task must match the complete normalized task oracle, which pins the existing-boundary trigger, file:line citation, covering test input, and production producer together and rejects reversal, omission, or weakening.", + "timestamp": "2026-08-02T23:11:02.928Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.2--df4ea1367363.json b/.2119/verdicts/self-supplied-evidence.2.2--df4ea1367363.json new file mode 100644 index 0000000..3b10900 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.2--df4ea1367363.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.2--df4ea1367363", + "requirementId": "self-supplied-evidence.2.2", + "hash": "df4ea1367363", + "verdict": "pass", + "summary": "The independent full-block oracle now pins 'If that boundary exists' together with file:line citation, test input, and that producer for every production-dispatched test-quality instruction, rejecting reversed or weakened triggers.", + "timestamp": "2026-08-02T23:08:52.297Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.3--3bfc1f157e4c.json b/.2119/verdicts/self-supplied-evidence.2.3--3bfc1f157e4c.json new file mode 100644 index 0000000..7b8eaec --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.3--3bfc1f157e4c.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.3--3bfc1f157e4c", + "requirementId": "self-supplied-evidence.2.3", + "hash": "3bfc1f157e4c", + "verdict": "pass", + "summary": "The generated-task test covers every production-derived test-quality target and rejects omission or weakening of the required file:line trace that the exercised input preserves the production producer's value shape.", + "timestamp": "2026-08-02T05:33:56.135Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.3--a0e14d8db077.json b/.2119/verdicts/self-supplied-evidence.2.3--a0e14d8db077.json new file mode 100644 index 0000000..8d901aa --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.3--a0e14d8db077.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.3--a0e14d8db077", + "requirementId": "self-supplied-evidence.2.3", + "hash": "a0e14d8db077", + "verdict": "pass", + "summary": "Every production-discovered test-quality task is full-body matched, pinning the conditional file:line demand that the covering test's exercised input preserve the production producer's value shape.", + "timestamp": "2026-08-02T23:14:29.592Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.3--b768f1b9b787.json b/.2119/verdicts/self-supplied-evidence.2.3--b768f1b9b787.json new file mode 100644 index 0000000..4271f08 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.3--b768f1b9b787.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.3--b768f1b9b787", + "requirementId": "self-supplied-evidence.2.3", + "hash": "b768f1b9b787", + "verdict": "pass", + "summary": "Every production-discovered test-quality instruction is full-body matched, including the conditional file:line demand that the exercised value preserve the production producer's shape; omissions, weakenings, and contradictions fail.", + "timestamp": "2026-08-02T23:11:13.182Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.4--2755a46ec65c.json b/.2119/verdicts/self-supplied-evidence.2.4--2755a46ec65c.json new file mode 100644 index 0000000..1382bcc --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.4--2755a46ec65c.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.4--2755a46ec65c", + "requirementId": "self-supplied-evidence.2.4", + "hash": "2755a46ec65c", + "verdict": "pass", + "summary": "Fresh dispatch integration verifies every generated test-quality task explicitly requires file:line evidence distinguishing a newly produced observation from an equal pre-existing sentinel, while shared rejection checks forbid weakening language.", + "timestamp": "2026-08-02T05:34:49.897Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.4--2e7bb01145d0.json b/.2119/verdicts/self-supplied-evidence.2.4--2e7bb01145d0.json new file mode 100644 index 0000000..bcb1b12 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.4--2e7bb01145d0.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.4--2e7bb01145d0", + "requirementId": "self-supplied-evidence.2.4", + "hash": "2e7bb01145d0", + "verdict": "pass", + "summary": "Every computed production-generated test-quality task matches the complete oracle, pinning the decisive-observation trigger, all four initial/default/placeholder/sentinel cases, file:line evidence, and newly-produced versus pre-existing distinction.", + "timestamp": "2026-08-02T23:14:34.728Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.2.4--3add2f8a8874.json b/.2119/verdicts/self-supplied-evidence.2.4--3add2f8a8874.json new file mode 100644 index 0000000..076930f --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.2.4--3add2f8a8874.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.2.4--3add2f8a8874", + "requirementId": "self-supplied-evidence.2.4", + "hash": "3add2f8a8874", + "verdict": "pass", + "summary": "The complete task snapshot pins the conditional trigger, all four pre-existing-value classes, file:line evidence, and the newly-produced-versus-pre-existing distinction for every production-computed test-quality target.", + "timestamp": "2026-08-02T23:11:20.691Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.3.1--0c0cd9b96a10.json b/.2119/verdicts/self-supplied-evidence.3.1--0c0cd9b96a10.json new file mode 100644 index 0000000..67f8bab --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.3.1--0c0cd9b96a10.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.3.1--0c0cd9b96a10", + "requirementId": "self-supplied-evidence.3.1", + "hash": "0c0cd9b96a10", + "verdict": "pass", + "summary": "Every production-computed test-quality task exactly defines the boundary as invocation of a binary or service outside the gate's own process, with the complete task snapshot rejecting weakened definitions.", + "timestamp": "2026-08-02T23:14:48.392Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.3.1--8f4768f6eda9.json b/.2119/verdicts/self-supplied-evidence.3.1--8f4768f6eda9.json new file mode 100644 index 0000000..e061bc3 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.3.1--8f4768f6eda9.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.3.1--8f4768f6eda9", + "requirementId": "self-supplied-evidence.3.1", + "hash": "8f4768f6eda9", + "verdict": "pass", + "summary": "Enumerates every production-derived test-quality task and rejects missing or altered boundary wording, including inside-process and non-binary/service near-counterexamples, via an exact required definition.", + "timestamp": "2026-08-02T05:36:17.365Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.3.1--ba8be2d406ca.json b/.2119/verdicts/self-supplied-evidence.3.1--ba8be2d406ca.json new file mode 100644 index 0000000..28ef528 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.3.1--ba8be2d406ca.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.3.1--ba8be2d406ca", + "requirementId": "self-supplied-evidence.3.1", + "hash": "ba8be2d406ca", + "verdict": "pass", + "summary": "Every production-generated test-quality task must match the complete normalized oracle, so changing invocation, binary-or-service scope, or outside-the-gate-process boundary wording fails.", + "timestamp": "2026-08-02T23:11:23.747Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.3.2--401f4b68deae.json b/.2119/verdicts/self-supplied-evidence.3.2--401f4b68deae.json new file mode 100644 index 0000000..37faad6 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.3.2--401f4b68deae.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.3.2--401f4b68deae", + "requirementId": "self-supplied-evidence.3.2", + "hash": "401f4b68deae", + "verdict": "pass", + "summary": "Every production-discovered test-quality task is full-body matched, pinning the conditional file:line demand for both the dependency's production provisioning declaration and its production absence-failure path.", + "timestamp": "2026-08-02T23:14:51.665Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.3.2--5f8200b2fdd0.json b/.2119/verdicts/self-supplied-evidence.3.2--5f8200b2fdd0.json new file mode 100644 index 0000000..2d27223 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.3.2--5f8200b2fdd0.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.3.2--5f8200b2fdd0", + "requirementId": "self-supplied-evidence.3.2", + "hash": "5f8200b2fdd0", + "verdict": "pass", + "summary": "PASS: every generated test-quality task is checked for an exact conditional instruction requiring file:line evidence of both production provisioning and the production absence-failure path; omitting the boundary condition, either conjunct, or file:line demand fails.", + "timestamp": "2026-08-02T05:36:21.818Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.3.2--781809753ed3.json b/.2119/verdicts/self-supplied-evidence.3.2--781809753ed3.json new file mode 100644 index 0000000..b4400d7 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.3.2--781809753ed3.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.3.2--781809753ed3", + "requirementId": "self-supplied-evidence.3.2", + "hash": "781809753ed3", + "verdict": "pass", + "summary": "Every production-discovered test-quality instruction is full-body matched, pinning the conditional file:line demand for both the production provisioning declaration and the production path that fails when the external dependency is absent.", + "timestamp": "2026-08-02T23:11:33.535Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.4.1--02d88bd20370.json b/.2119/verdicts/self-supplied-evidence.4.1--02d88bd20370.json new file mode 100644 index 0000000..33fce7f --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.4.1--02d88bd20370.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.4.1--02d88bd20370", + "requirementId": "self-supplied-evidence.4.1", + "hash": "02d88bd20370", + "verdict": "pass", + "summary": "Every computed production-generated test-quality task matches the complete oracle, pinning a FAIL for either absent applicable provenance or evidence that production cannot produce the claimed failure independently of test setup.", + "timestamp": "2026-08-02T23:14:55.202Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.4.1--38564acf58d7.json b/.2119/verdicts/self-supplied-evidence.4.1--38564acf58d7.json new file mode 100644 index 0000000..1bb792a --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.4.1--38564acf58d7.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.4.1--38564acf58d7", + "requirementId": "self-supplied-evidence.4.1", + "hash": "38564acf58d7", + "verdict": "pass", + "summary": "The complete task snapshot requires Record FAIL for both absent applicable provenance evidence and evidence that production cannot produce the failure independently of test setup across every production-computed test-quality target.", + "timestamp": "2026-08-02T23:11:40.674Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.4.1--dbb07998ba99.json b/.2119/verdicts/self-supplied-evidence.4.1--dbb07998ba99.json new file mode 100644 index 0000000..1b4c3b3 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.4.1--dbb07998ba99.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.4.1--dbb07998ba99", + "requirementId": "self-supplied-evidence.4.1", + "hash": "dbb07998ba99", + "verdict": "pass", + "summary": "Generated instructions are checked across every production-derived test-quality target and must explicitly record FAIL for either missing provenance or evidence that production cannot independently produce the failure; omission, weakened wording, and escape-clause counterexamples are rejected.", + "timestamp": "2026-08-02T05:36:27.004Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.5.1--15d76cbfb0f1.json b/.2119/verdicts/self-supplied-evidence.5.1--15d76cbfb0f1.json new file mode 100644 index 0000000..32d7281 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.5.1--15d76cbfb0f1.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.5.1--15d76cbfb0f1", + "requirementId": "self-supplied-evidence.5.1", + "hash": "15d76cbfb0f1", + "verdict": "fail", + "summary": "The test inspects only REQ-003.4.2 instead of every computed direct-judgment target; adding provenance questions only to another generated [review] instruction remains green, and rephrased provenance questions can evade the phrase blacklist.", + "timestamp": "2026-08-02T23:12:01.145Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.5.1--4bf3f2230530.json b/.2119/verdicts/self-supplied-evidence.5.1--4bf3f2230530.json new file mode 100644 index 0000000..abfbe7a --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.5.1--4bf3f2230530.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.5.1--4bf3f2230530", + "requirementId": "self-supplied-evidence.5.1", + "hash": "4bf3f2230530", + "verdict": "pass", + "summary": "Every production-computed direct-judgment target is compared with the complete canonical direct task, which contains no test-quality provenance questionnaire and fails on any insertion.", + "timestamp": "2026-08-02T23:15:24.252Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.5.1--e7eb54958e86.json b/.2119/verdicts/self-supplied-evidence.5.1--e7eb54958e86.json new file mode 100644 index 0000000..0c79deb --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.5.1--e7eb54958e86.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.5.1--e7eb54958e86", + "requirementId": "self-supplied-evidence.5.1", + "hash": "e7eb54958e86", + "verdict": "pass", + "summary": "The test exercises a real generated [review] direct-judgment instruction and rejects inclusion of every specified test-quality provenance question category while confirming the direct-judgment prompt remains present.", + "timestamp": "2026-08-02T05:38:41.022Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.6.1--140a2a47aa19.json b/.2119/verdicts/self-supplied-evidence.6.1--140a2a47aa19.json new file mode 100644 index 0000000..e1fc4f6 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.6.1--140a2a47aa19.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.6.1--140a2a47aa19", + "requirementId": "self-supplied-evidence.6.1", + "hash": "140a2a47aa19", + "verdict": "pass", + "summary": "PASS: the real lint workflow exits clean on a covering test containing both literal config input and a factory-built fixture, so either syntax-triggered provenance violation would fail the test.", + "timestamp": "2026-08-02T21:27:52.554Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.6.1--68dbeff8af0f.json b/.2119/verdicts/self-supplied-evidence.6.1--68dbeff8af0f.json new file mode 100644 index 0000000..fe7c9b3 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.6.1--68dbeff8af0f.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.6.1--68dbeff8af0f", + "requirementId": "self-supplied-evidence.6.1", + "hash": "68dbeff8af0f", + "verdict": "pass", + "summary": "The production lint CLI scans copied annotated tests containing both literal constructions and factory functions, and the test requires status 0 plus the exact clean-spec count, so a syntax-only provenance violation fails.", + "timestamp": "2026-08-02T23:12:28.571Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.6.1--978277912461.json b/.2119/verdicts/self-supplied-evidence.6.1--978277912461.json new file mode 100644 index 0000000..1a3721e --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.6.1--978277912461.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.6.1--978277912461", + "requirementId": "self-supplied-evidence.6.1", + "hash": "978277912461", + "verdict": "pass", + "summary": "The production lint CLI scans an otherwise clean fixture whose copied covering tests contain both literal construction and factory syntax; any provenance violation inferred solely from either syntax makes the asserted zero status fail.", + "timestamp": "2026-08-02T23:15:15.194Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.7.1--5676f3ef50cb.json b/.2119/verdicts/self-supplied-evidence.7.1--5676f3ef50cb.json new file mode 100644 index 0000000..54a9a8b --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.7.1--5676f3ef50cb.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.7.1--5676f3ef50cb", + "requirementId": "self-supplied-evidence.7.1", + "hash": "5676f3ef50cb", + "verdict": "pass", + "summary": "All mechanically computed standard targets match complete kind-specific task oracles, and audit filenames exactly equal the computed passing-target set with every audit matching its full oracle; each pins concrete names, cardinality, and the category-promotion prohibition.", + "timestamp": "2026-08-02T23:15:20.011Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.7.1--a75c8b3b6ac2.json b/.2119/verdicts/self-supplied-evidence.7.1--a75c8b3b6ac2.json new file mode 100644 index 0000000..3ff6975 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.7.1--a75c8b3b6ac2.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.7.1--a75c8b3b6ac2", + "requirementId": "self-supplied-evidence.7.1", + "hash": "a75c8b3b6ac2", + "verdict": "fail", + "summary": "Standard and audit instructions are covered, but contradictory guidance such as 'A category-level summary is sufficient for member-specific evidence' bypasses the negative regex and leaves every assertion green.", + "timestamp": "2026-08-02T23:12:16.515Z" +} diff --git a/.2119/verdicts/self-supplied-evidence.7.1--ae376b3595ab.json b/.2119/verdicts/self-supplied-evidence.7.1--ae376b3595ab.json new file mode 100644 index 0000000..18ce592 --- /dev/null +++ b/.2119/verdicts/self-supplied-evidence.7.1--ae376b3595ab.json @@ -0,0 +1,8 @@ +{ + "reviewId": "self-supplied-evidence.7.1--ae376b3595ab", + "requirementId": "self-supplied-evidence.7.1", + "hash": "ae376b3595ab", + "verdict": "pass", + "summary": "The fixture rejects missing or weakened evidence-bounded wording for every computed standard target and exactly every audit generated from its computed passing-verdict targets.", + "timestamp": "2026-08-02T21:30:44.081Z" +} diff --git a/eval/calibration/012-pass-control-perturbation.md b/eval/calibration/012-pass-control-perturbation.md index f2fb293..e1e6ee8 100644 --- a/eval/calibration/012-pass-control-perturbation.md +++ b/eval/calibration/012-pass-control-perturbation.md @@ -13,6 +13,12 @@ failure_mode: none — control case ## Evidence +Production provenance: this condensed case is taken from +`tests/review.test.ts:84`, with the independent perturbation assertions at +`tests/review.test.ts:94`, `tests/review.test.ts:100`, and +`tests/review.test.ts:104`. Those lines invoke the production review-ID +computation through `gearTargets` (`tests/review.test.ts:78`). + ```ts // 2119: FIX-001.1.1 it("scopes review IDs to evidence blocks", () => { diff --git a/eval/calibration/013-pass-control-negative-positive.md b/eval/calibration/013-pass-control-negative-positive.md index 0fc1837..6ee21fb 100644 --- a/eval/calibration/013-pass-control-negative-positive.md +++ b/eval/calibration/013-pass-control-negative-positive.md @@ -7,28 +7,31 @@ failure_mode: none — control case ## Requirement -> A record MUST be counted only when fully well-formed: `kind` exactly `a` or `b`, nonempty -> `note`, `owner` consistent with the ID, and a parseable `stamp`; anything else is a loud -> violation naming the file, never a silent pass. +> A verdict record MUST be rejected loudly when its `verdict` is absent or is not exactly `pass` +> or `fail`, while a record written by the real CLI is accepted. ## Evidence +Production provenance: this case condenses `tests/verdict-validation.test.ts:47-75`. The +historical bare-record failure is exercised at lines 51-55, the near-miss `passd` verdict at +lines 57-71, and the real CLI writer control at lines 73-75. + ```ts // 2119: FIX-001.1.1 -it("counts only well-formed records, loudly rejecting the rest", () => { - write({ id }); // bare record: the reported exploit +it("counts a verdict only when the full record is well-formed", () => { + write({ reviewId }); // bare record: the reported exploit + expect(check().status).toBe(1); + expect(check().stderr).toContain("malformed verdict file"); + write({ reviewId, requirementId, hash, verdict: "passd", summary, timestamp }); expect(check().status).toBe(1); - expect(check().stderr).toContain("malformed"); - write({ id, kind: "banana", note: "n", owner, stamp }); // typo'd kind - expect(check().stderr).toContain('exactly "a" or "b"'); - writeViaCli(id, "genuine note"); // well-formed via the real writer + expect(check().stderr).toContain('verdict must be exactly "pass" or "fail"'); + expect(runCli("pass", reviewId, "--summary", "asserts spin").status).toBe(0); expect(check().status).toBe(0); }); ``` ## Why the correct verdict is PASS -Negative and positive controls both present: the historical exploit is reproduced verbatim and -rejected, a near-miss (typo'd enum) is rejected with the named reason, and the genuine record is -accepted — proving the gate fails closed without over-rejecting. Rejection *reasons* are -asserted, so the violation is loud, not silent, satisfying that boundary term too. +Negative and positive controls are all production-backed: the historical missing-verdict exploit +is rejected, a near-miss verdict value is rejected with the named reason, and the CLI-written +record is accepted. The gate therefore fails closed without over-rejecting. diff --git a/eval/calibration/014-scope-inflated-verdict.md b/eval/calibration/014-scope-inflated-verdict.md new file mode 100644 index 0000000..0759ec2 --- /dev/null +++ b/eval/calibration/014-scope-inflated-verdict.md @@ -0,0 +1,29 @@ +--- +expected_verdict: fail +prompt: test-quality +source: panopticon PR #112 follow-up review, 2026-08-02 +failure_mode: scope inflation — singular evidence promoted into a plural category claim +--- + +## Requirement + +> Task-service shell callers MUST authenticate their requests. + +## Evidence + +```sh +# 2119: FIX-001.1.1 +test_task_lib_authenticates() { + run_shell_with_recording_curl shell/task_lib.sh + assert_recorded_argv_contains "Authorization: Bearer" +} +``` + +Candidate verdict summary: “The tests exercise authenticated shell-library requests.” + +## Why the correct verdict is FAIL + +The recording-stub technique honestly observes `shell/task_lib.sh`, but the candidate verdict +promotes that one named caller into plural “shell-library requests.” Its subject is broader than +its cited evidence. The summary must name `shell/task_lib.sh`; establishing the whole category +requires separate mechanical enumeration of the real caller set. diff --git a/specs/self-supplied-evidence.md b/specs/self-supplied-evidence.md new file mode 100644 index 0000000..ff7a72c --- /dev/null +++ b/specs/self-supplied-evidence.md @@ -0,0 +1,122 @@ +# Self-Supplied Evidence in Judgment Reviews + +## Overview + +Test-quality reviews can approve a passing test whose decisive evidence was supplied by the +checker, fixture, or prompt rather than by the production system. Such a test is internally +consistent but cannot establish the production property: a fixture may invent a field the real +producer omits, a test may emit the signal its watcher is meant to observe, a prompt may prescribe +the identity a report claims, an undeclared tool may exist only because the live agent installed +it, or an initial sentinel may overlap a real runtime value. + +This feature changes the generated, per-requirement test-quality review instructions. It does not +try to infer boundary crossings from requirement keywords: that would recreate the project's +keyword-grep anti-pattern and still miss implicit boundaries. Instead every reviewer answers one +short independence question, and only reviewers who identify a producer/consumer or +gate/environment boundary pay for a second question. Both answers demand `file:line` evidence so +the decision remains checkable in fresh context. + +The candidate rules were evaluated as follows: + +- **Provenance of inputs survives, conditionally.** At a producer/consumer boundary, an honest + test must exercise an input the production producer can actually emit. Requiring this of every + test would burden pure functions and internal state transitions, so it is activated only when + the reviewer identifies that boundary. +- **Provenance of claims survives in generalized form.** A self-report, readiness signal, cache + state, or other decisive observation must be independent of the test setup or parameter that + asks for the claim. This is not limited to systems literally describing themselves; the broader + independence question also catches signal injection and sentinel/value collisions. +- **Declared environment survives, conditionally.** A test of a gate with an external binary or + service must connect the dependency to its production provisioning declaration and to the + production absence failure. Like input provenance, this applies only when the reviewer finds + such a boundary. +- **Literal-input lint is rejected for now.** Literals are legitimate at many boundaries, while + factory-built objects can be equally fictional. Syntax alone cannot decide provenance. The + generated reviewer can trace values across `file:line` evidence. Viable future tooling has two + shapes: use explicit machine-readable producer metadata, or enumerate the real artifact set + (for example, routes on the composed app, shell files that reference a service URL, or registered + harnesses) and assert the property over every discovered member. Such a sweep must treat an empty + result as an error; a silently empty glob would merely reproduce self-supplied evidence inside + the lint. Neither approach should guess provenance from literal or factory syntax. + +Later fleet evidence exposed a neighboring failure class: **scope inflation**. One honest test +captured real runtime arguments for one shell library, but its committed verdict claimed that +plural “shell libraries” were exercised; two unenumerated callers included a live security defect. +The provenance questions in this spec intentionally do not claim to catch that case: production +can reach the tested failure, and the test's evidence for that one caller is independent and real. +What is false is the judgment's generalization from one set member to the category. A conditional +question asking a reviewer to enumerate plural or categorical subjects is the wrong instrument: +the reviewer's enumeration would itself be an unverified claim capable of the same generalization +error. Reliable scope completeness requires tooling that mechanically enumerates the real artifact +and fails when enumeration returns no members. That separate commitment is tracked in +[issue #17](https://github.com/Unsupervisedcom/2119/issues/17). Adding it here would also lengthen +every test-quality review with a third concern that is neither input provenance nor self-supplied +evidence. Recording the boundary prevents a future author from assuming this feature covers scope +inflation or adding the analytical question casually. + +The motivating verdict also contained a cheaper, locally decidable wording defect: evidence for +the named file `task_lib.sh` was summarized as coverage of plural “shell libraries.” Preventing that +promotion does not require discovering the full set. This spec therefore constrains verdict-writing +guidance, not the review analysis: a conclusion preserves the concrete subject names and cardinality +of the evidence it cites. The rule applies to both test-quality and direct-judgment instructions, +adds no question, and does not alter the independence or conditional-boundary decisions above. It +cannot establish category completeness—that remains issue #17—but it prevents a verdict from +claiming category completeness when its own cited evidence is only member-specific. + +Two additional auth failures do fall within this spec. A test that merely found a credential +template string, instead of executing the shell and capturing its arguments, cannot satisfy the +production-reachability and independent-observation obligation in 1.2. Tests that called +`create_app()` only with explicit keyword arguments, while production used environment variables +through `build_app()`, cannot satisfy 2.2's requirement to obtain boundary input through the real +production producer; the unexercised fail-open default is precisely the production failure 1.1 +requires the reviewer to name. + +The cheap proposed question — name the production failure and confirm production can produce it — +is the universal core. The added conditional questions make “confirm” decidable. A +producer/consumer boundary exists here only when the behavior consumes a value emitted by a +separately invoked production component or production data source; an ordinary call between units +inside the behavior under test is not enough. A runtime-environment boundary exists when the gate +invokes a binary or service outside its own process. These definitions trigger focused provenance +traces without making every pure unit test pay the cost. + +This feature's own acceptance tests invoke the built CLI's real `review --dispatch` workflow +against this checked-in file-scoped spec and its annotated tests, then inspect instructions that +workflow actually writes. They do not satisfy coverage by directly calling the instruction +renderer with a hand-built requirement or by asserting against a separately copied prompt fixture. +Thus the production parser, coverage resolver, review-target computation, and instruction writer +supply the artifacts under assertion. + +## Requirements + +### 1: Independent production failure + +1. Each generated test-quality review instruction MUST require the reviewer to name a concrete production failure the covering test would catch. +2. Each generated test-quality review instruction MUST require `file:line` evidence that production can reach the named failure without the test, its fixtures, or its prompts supplying the triggering input or decisive observation. + +### 2: Conditional boundary provenance + +1. Each generated test-quality review instruction MUST define a producer/consumer boundary as consumption of a value emitted by a separately invoked production component or production data source. +2. Each generated test-quality review instruction MUST require the reviewer, whenever a producer/consumer boundary exists, to cite `file:line` evidence that the covering test obtains its input through that production producer. +3. Each generated test-quality review instruction MUST require the reviewer, whenever a producer/consumer boundary exists, to cite `file:line` evidence that the input exercised by the covering test preserves the production producer's value shape. +4. Each generated test-quality review instruction MUST require the reviewer, whenever the decisive observation can equal an initial, default, placeholder, or sentinel value, to cite `file:line` evidence that the covering test distinguishes a newly produced observation from that pre-existing value. + +### 3: Declared runtime environment + +1. Each generated test-quality review instruction MUST define a gate/runtime-environment boundary as invocation of a binary or service outside the gate's own process. +2. Each generated test-quality review instruction MUST require the reviewer, whenever a gate/runtime-environment boundary exists, to cite `file:line` evidence of both the dependency's production provisioning declaration and the production path that fails when the dependency is absent. + +### 4: Decidable verdicts + +1. Each generated test-quality review instruction MUST direct the reviewer to fail the judgment when any provenance evidence required by its applicable questions is absent or shows that production cannot produce the claimed failure independently of the test setup. + +### 5: Scope + +1. Generated direct-judgment instructions for `[review]` requirements MUST remain exempt from the test-quality provenance questions. + +### 6: Lint compatibility + +1. Requirement linting MUST NOT report a provenance violation solely because a covering test constructs an input with literal or factory syntax. + +### 7: Evidence-bounded verdict wording + +1. Each generated review instruction MUST direct the reviewer to keep the verdict summary's subject no broader than the cited evidence, preserving concrete member names and singular or plural scope instead of promoting member-specific evidence into a category claim. diff --git a/src/review.ts b/src/review.ts index f5352a7..0c890b8 100644 --- a/src/review.ts +++ b/src/review.ts @@ -194,6 +194,8 @@ implementation's current behavior. ## Recording your verdict +Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim. + \`\`\` npx rfc2119 pass ${t.reviewId} --summary "audit: " npx rfc2119 fail ${t.reviewId} --summary "audit: " @@ -242,6 +244,20 @@ Read the requirement and each evidence file's tests annotated with \`2119: ${t.r - **Unrelated assertions** — tests that reference the requirement ID but assert something other than its criterion. - **Keyword theater** — string/keyword matching standing in for behavioral verification. +**Required production-provenance answers (a PASS is forbidden without them):** + +1. Name the concrete production failure this test would catch. + Cite file:line evidence that production can reach that failure without the test, fixtures, or prompts supplying the trigger or decisive observation. +2. Trace each applicable production boundary with file:line evidence. + A producer/consumer boundary means consuming a value emitted by a separately invoked production component or production data source. + If that boundary exists, cite file:line evidence that the test obtains its input from that producer. + If that boundary exists, cite file:line evidence that the exercised value preserves the producer's production shape. + If the decisive observation can equal an initial/default/placeholder/sentinel value, cite file:line evidence that the test distinguishes a newly produced observation from that pre-existing value. + A gate/runtime-environment boundary means invoking a binary or service outside the gate's own process. + If that boundary exists, cite file:line evidence for both its production provisioning declaration and the production path that fails when it is absent. + +Record FAIL when applicable provenance evidence is absent or shows that production cannot produce the failure independently of the test setup. + **Counterexample obligation:** enumerate the requirement's conjuncts and boundary terms (words like "comment", "exactly", "only", "begins with"). For each, construct the nearest violating input — the almost-conforming case the requirement forbids — and confirm a test rejects it. @@ -290,6 +306,8 @@ requirement honestly tested is still a bad requirement. ${custom ? `\n## Additional review criteria\n\n*(from \`${custom.path}\` — these extend the requirement above)*\n\n${custom.content}\n` : ""} ## Recording your verdict +Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim. + If the requirement's verification is genuine (or all findings were fixed), run: \`\`\` diff --git a/tests/hook.test.ts b/tests/hook.test.ts index 900aacf..8807b39 100644 --- a/tests/hook.test.ts +++ b/tests/hook.test.ts @@ -85,13 +85,46 @@ describe("normalized hook handling", () => { // 2119: REQ-004.1.7 it("injects workflow context on session-start", async () => { - const root = fixture(); - const out = await handleHook(root, "session-start", "claude", {}) as { - hookSpecificOutput?: { hookEventName: string; additionalContext: string }; - }; - expect(out.hookSpecificOutput?.hookEventName).toBe("SessionStart"); - expect(out.hookSpecificOutput?.additionalContext).toBe(SESSION_CONTEXT); - expect(SESSION_CONTEXT).toContain("npx rfc2119 check"); + const { readFileSync } = await import("node:fs"); + const { spawnSync } = await import("node:child_process"); + const { installAgentHooks } = await import("../src/adapters.js"); + for (const platform of ["claude", "codex", "gemini"] as const) { + const root = fixture(); + const cache = join(root, "upgrade.json"); + writeFileSync(cache, JSON.stringify({ checkedAt: Date.now(), latest: "0.0.0" })); + process.env.RFC2119_UPGRADE_CACHE = cache; + try { + installAgentHooks(root, platform); + const settingsPath = platform === "claude" + ? join(root, ".claude/settings.json") + : platform === "codex" + ? join(root, ".codex/hooks.json") + : join(root, ".gemini/settings.json"); + const settings = JSON.parse(readFileSync(settingsPath, "utf8")); + const command = settings.hooks.SessionStart[0].hooks[0].command as string; + expect(command).toContain(`hook session-start --platform ${platform}`); + + const run = spawnSync(command, { + cwd: root, + input: "{}", + encoding: "utf8", + env: process.env, + shell: true, + }); + expect(run.status).toBe(0); + const out = JSON.parse(run.stdout) as { + hookSpecificOutput?: { hookEventName: string; additionalContext: string }; + }; + expect(out.hookSpecificOutput?.hookEventName).toBe("SessionStart"); + const context = out.hookSpecificOutput?.additionalContext ?? ""; + expect(context).toContain("spec-driven testing"); + expect(context).toContain("every MUST-level requirement"); + expect(context).toContain("npx rfc2119 check"); + expect(context.length).toBeLessThan(1000); + } finally { + delete process.env.RFC2119_UPGRADE_CACHE; + } + } }); // 2119: REQ-004.1.8 diff --git a/tests/self-supplied-evidence.test.ts b/tests/self-supplied-evidence.test.ts new file mode 100644 index 0000000..0e36ca5 --- /dev/null +++ b/tests/self-supplied-evidence.test.ts @@ -0,0 +1,312 @@ +import { execFileSync } from "node:child_process"; +import { cpSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join, resolve } from "node:path"; +import { beforeAll, describe, expect, it } from "vitest"; +import { buildContext } from "../src/check.js"; + +const CLI = resolve(import.meta.dirname, "../dist/cli.js"); +const REPO = resolve(import.meta.dirname, ".."); +const EXPECTED_PROVENANCE = `1. Name the concrete production failure this test would catch. + Cite file:line evidence that production can reach that failure without the test, fixtures, or prompts supplying the trigger or decisive observation. +2. Trace each applicable production boundary with file:line evidence. + A producer/consumer boundary means consuming a value emitted by a separately invoked production component or production data source. + If that boundary exists, cite file:line evidence that the test obtains its input from that producer. + If that boundary exists, cite file:line evidence that the exercised value preserves the producer's production shape. + If the decisive observation can equal an initial/default/placeholder/sentinel value, cite file:line evidence that the test distinguishes a newly produced observation from that pre-existing value. + A gate/runtime-environment boundary means invoking a binary or service outside the gate's own process. + If that boundary exists, cite file:line evidence for both its production provisioning declaration and the production path that fails when it is absent. + +Record FAIL when applicable provenance evidence is absent or shows that production cannot produce the failure independently of the test setup.`; +const EXPECTED_TASK = `**Would the covering tests fail if this requirement were violated?** + +Read the requirement and each evidence file's tests annotated with \`2119: \` (or its section ID). Judge whether they genuinely verify the requirement. You MUST flag: + +- **Tautological assertions** — tests that assert what they just set up, or that cannot fail. +- **Over-mocking** — mocks/stubs that bypass the very behavior the requirement constrains. +- **Unrelated assertions** — tests that reference the requirement ID but assert something other than its criterion. +- **Keyword theater** — string/keyword matching standing in for behavioral verification. + +**Required production-provenance answers (a PASS is forbidden without them):** + +${EXPECTED_PROVENANCE} + +**Counterexample obligation:** enumerate the requirement's conjuncts and boundary terms (words +like "comment", "exactly", "only", "begins with"). For each, construct the nearest violating +input — the almost-conforming case the requirement forbids — and confirm a test rejects it. +Do not reason from the implementation's current behavior; reason from the requirement's text. +A review that cannot name a rejected counterexample for a boundary term is not a pass. + +**Judge the requirement too:** if the requirement itself is ambiguous, untestable, or states an +implementation mechanism rather than an observable outcome, fail with that finding — a bad +requirement honestly tested is still a bad requirement. + +## Recording your verdict + +Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim. + +If the requirement's verification is genuine (or all findings were fixed), run: + +\`\`\` +npx rfc2119 pass --summary "" +\`\`\` + +If there are unresolved findings, run: + +\`\`\` +npx rfc2119 fail --summary "" +\`\`\` + +The summary is committed to the repository and read by humans in PR review — +be specific. Do not edit any files; report, don't fix.`; +const RECORDING_GUIDANCE = "Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular/plural scope; do not promote member-specific evidence into a category claim."; +const EXPECTED_DIRECT_TASK = `**Is this requirement genuinely satisfied by the current state of the evidence files?** + +Read the requirement and the evidence files and judge compliance directly. This requirement was tagged \`[review]\` because it needs judgment rather than a test. + +**Judge the requirement too:** if the requirement itself is ambiguous, untestable, or states an +implementation mechanism rather than an observable outcome, fail with that finding — a bad +requirement honestly tested is still a bad requirement. + +## Recording your verdict + +${RECORDING_GUIDANCE} + +If the requirement's verification is genuine (or all findings were fixed), run: + +\`\`\` +npx rfc2119 pass --summary "" +\`\`\` + +If there are unresolved findings, run: + +\`\`\` +npx rfc2119 fail --summary "" +\`\`\` + +The summary is committed to the repository and read by humans in PR review — +be specific. Do not edit any files; report, don't fix.`; +const EXPECTED_AUDIT_TASK = `**Construct a concrete mutant or input under which this requirement is violated while every +covering test stays green.** Enumerate the requirement's conjuncts and boundary terms; probe the +negative space (what must be refused, not what is accepted); consider shared fixtures, preludes, +and paths the tests never touch. Reason from the requirement's text, never from the +implementation's current behavior. + +- If you find such a counterexample: record a FAIL with the mutant described concretely enough + to reproduce. +- Only if you genuinely cannot construct one after honest effort: record a PASS stating the + strongest candidate you tried and why it fails to survive. + +## Recording your verdict + +${RECORDING_GUIDANCE} + +\`\`\` +npx rfc2119 pass --summary "audit: " +npx rfc2119 fail --summary "audit: " +\`\`\` + +Do not edit any files; report, don't fix.`; + +function normalizedTask(body: string): string { + return body + .replace(/[A-Za-z][A-Za-z0-9-]*\.\d+\.\d+--[0-9a-f]{12}/g, "") + .replace(/[A-Za-z][A-Za-z0-9-]*\.\d+\.\d+/g, "") + .trim(); +} + +// Bare annotations below resolve through the real file-scoped spec copied by dispatchFixture(). +// 2119-spec: self-supplied-evidence + +function run(cwd: string, args: string[]): { status: number; stdout: string } { + try { + return { status: 0, stdout: execFileSync("node", [CLI, ...args], { cwd, encoding: "utf8" }) }; + } catch (err) { + const e = err as { status: number; stdout: string }; + return { status: e.status, stdout: e.stdout ?? "" }; + } +} + +function dispatchFixture(): string { + const root = realpathSync(mkdtempSync(join(tmpdir(), "2119-self-evidence-"))); + mkdirSync(join(root, "specs")); + mkdirSync(join(root, "tests")); + mkdirSync(join(root, "src")); + cpSync(join(REPO, "specs/self-supplied-evidence.md"), join(root, "specs/self-supplied-evidence.md")); + cpSync(join(REPO, "tests/self-supplied-evidence.test.ts"), join(root, "tests/self-supplied-evidence.test.ts")); + cpSync(join(REPO, "tests/dispatch.test.ts"), join(root, "tests/dispatch.test.ts")); + + // A real checked-in [review] requirement supplies the direct-judgment control. + cpSync(join(REPO, "specs/REQ-003-judgment-reviews.md"), join(root, "specs/REQ-003-judgment-reviews.md")); + cpSync(join(REPO, "README.md"), join(root, "README.md")); + cpSync(join(REPO, "src/review.ts"), join(root, "src/review.ts")); + writeFileSync( + join(root, ".2119.yml"), + 'specs: ["specs/**/*.md"]\ntests: ["tests/**"]\nprefix: "REQ"\nreview_model: "test-model"\n', + ); + expect(run(root, ["review", "--dispatch"]).status).toBe(1); + return root; +} + +function testQualityTaskBodies(root: string): string[] { + // Production parsing and coverage—not generated wording—identify the complete target set. + const targets = buildContext(root).reviewTargets.filter((target) => target.kind === "test-quality"); + const generated = readdirSync(join(root, ".2119/reviews")).filter((entry) => entry.endsWith(".md")); + return targets.map((target) => { + const matches = generated.filter((entry) => entry === `${target.reviewId}.md`); + expect(matches).toHaveLength(1); + return readFileSync(join(root, ".2119/reviews", matches[0]), "utf8").split("## Your task\n\n", 2)[1]; + }); +} + +function expectEveryTestQualityTask(root: string, assertion: (body: string) => void): void { + const bodies = testQualityTaskBodies(root); + expect(bodies.length).toBeGreaterThan(2); + for (const body of bodies) { + expect(normalizedTask(body)).toBe(EXPECTED_TASK); + const provenance = body + .split("**Required production-provenance answers (a PASS is forbidden without them):**", 2)[1] + ?.split("**Counterexample obligation:**", 1)[0]; + expect(provenance?.trim()).toBe(EXPECTED_PROVENANCE); + assertion(body); + } +} + +describe("self-supplied evidence review instructions", () => { + let root: string; + + beforeAll(() => { + root = dispatchFixture(); + }); + + // 2119: 1.1 + it("asks for the concrete production failure", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toMatch(/^1\. Name the concrete production failure this test would catch\.$/m); + }); + }); + + // 2119: 1.2 + it("demands a file:line production reachability trace independent of test setup", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toMatch( + /^ Cite file:line evidence that production can reach that failure without the test, fixtures, or prompts supplying the trigger or decisive observation\.$/m, + ); + }); + }); + + // 2119: 2.1 + it("defines the producer/consumer boundary narrowly", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toMatch( + /^ A producer\/consumer boundary means consuming a value emitted by a separately invoked production component or production data source\.$/m, + ); + }); + }); + + // 2119: 2.2 + it("requires applicable tests to source input from the production producer", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toContain("cite file:line evidence that the test obtains its input from that producer"); + }); + }); + + // 2119: 2.3 + it("requires applicable tests to preserve the producer's value shape", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toContain("cite file:line evidence that the exercised value preserves the producer's production shape"); + }); + }); + + // 2119: 2.4 + it("requires a new observation to be distinguished from pre-existing sentinels", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toMatch( + /^ If the decisive observation can equal an initial\/default\/placeholder\/sentinel value, cite file:line evidence that the test distinguishes a newly produced observation from that pre-existing value\.$/m, + ); + }); + }); + + // 2119: 3.1 + it("defines the runtime-environment boundary narrowly", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toMatch( + /^ A gate\/runtime-environment boundary means invoking a binary or service outside the gate's own process\.$/m, + ); + }); + }); + + // 2119: 3.2 + it("requires provisioning and absence-path evidence for external dependencies", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toMatch( + /^ If that boundary exists, cite file:line evidence for both its production provisioning declaration and the production path that fails when it is absent\.$/m, + ); + }); + }); + + // 2119: 4.1 + it("makes missing or self-supplied provenance a failing verdict", () => { + expectEveryTestQualityTask(root, (body) => { + expect(body).toContain("Record FAIL when applicable provenance evidence is absent or shows that production cannot produce the failure independently of the test setup."); + }); + }); + + // 2119: 5.1 + it("keeps the test-quality provenance questions out of direct judgments", () => { + const targets = buildContext(root).reviewTargets.filter((target) => target.kind === "requirement"); + expect(targets.length).toBeGreaterThan(0); + for (const target of targets) { + const direct = readFileSync(join(root, ".2119/reviews", `${target.reviewId}.md`), "utf8"); + expect(normalizedTask(direct.split("## Your task\n\n", 2)[1])).toBe(EXPECTED_DIRECT_TASK); + } + }); + + // 2119: 6.1 + it("does not infer provenance lint failures from literal or factory syntax", () => { + // The copied, annotated test contains both literal config text and a factory function; + // lint consumes those real files and must remain syntax-agnostic. + const result = run(root, ["lint"]); + expect(result.status).toBe(0); + expect(result.stdout).toContain("2 spec file(s) clean"); + }); + + // 2119: 7.1 + it("bounds every standard fixture target and every audit generated from committed verdicts", () => { + const expectBoundedGuidance = (body: string): void => { + const recordingGuidance = body.split("## Recording your verdict\n\n", 2)[1]; + expect(recordingGuidance).toMatch( + /^Keep the verdict summary's subject no broader than the cited evidence: preserve concrete member names and singular\/plural scope; do not promote member-specific evidence into a category claim\.$/m, + ); + expect(recordingGuidance).not.toMatch( + /ignore|disregard|optional|need not|not required|may (?:broaden|generalize|promote)|broader (?:scope|category) (?:is|remains) allowed/i, + ); + }; + + const targets = buildContext(root).reviewTargets; + expect(targets.length).toBeGreaterThan(13); + for (const target of targets) { + const standard = readFileSync(join(root, ".2119/reviews", `${target.reviewId}.md`), "utf8"); + expectBoundedGuidance(standard); + expect(normalizedTask(standard.split("## Your task\n\n", 2)[1])).toBe( + target.kind === "test-quality" ? EXPECTED_TASK : EXPECTED_DIRECT_TASK, + ); + } + + cpSync(join(REPO, ".2119/verdicts"), join(root, ".2119/verdicts"), { recursive: true }); + const withVerdicts = buildContext(root); + const passingTargets = withVerdicts.reviewTargets.filter( + (target) => withVerdicts.verdicts.get(target.reviewId)?.verdict === "pass", + ); + expect(passingTargets.length).toBeGreaterThan(2); + expect(run(root, ["review", "--audit", "--dispatch"]).status).toBe(1); + + const auditNames = readdirSync(join(root, ".2119/reviews")).filter((name) => name.endsWith(".audit.md")); + expect(auditNames.sort()).toEqual(passingTargets.map((target) => `${target.reviewId}.audit.md`).sort()); + for (const auditName of auditNames) { + const audit = readFileSync(join(root, ".2119/reviews", auditName), "utf8"); + expectBoundedGuidance(audit); + expect(normalizedTask(audit.split("## Your task\n\n", 2)[1])).toBe(EXPECTED_AUDIT_TASK); + } + }); +});