diff --git a/.perry/events.jsonl b/.perry/events.jsonl index 9c388f92..cef27393 100644 --- a/.perry/events.jsonl +++ b/.perry/events.jsonl @@ -579,3 +579,15 @@ {"ts": "2026-08-20T23:35:10", "event": "evidence", "id": "TASK-122", "title": "the repair path the tools advertise leaves the file needing a whitespace fix", "track": "main", "actor": "agent", "from": "—", "to": "evidence/2026-08/TASK-122-spec.md"} {"ts": "2026-08-20T23:37:40", "event": "status", "id": "TASK-140", "title": "every mode contract slot is assigned to an axis, and the spine-to-unit map is written down", "track": "main", "actor": "agent", "depends_on": [], "from": "in_progress", "to": "review", "reason": ""} {"ts": "2026-08-20T23:37:40", "event": "evidence", "id": "TASK-140", "title": "every mode contract slot is assigned to an axis, and the spine-to-unit map is written down", "track": "main", "actor": "agent", "from": "evidence/2026-08/TASK-140-dispatch-2026-08-21.md", "to": "evidence/2026-08/TASK-140-dispatch-2026-08-21-result.md"} +{"ts": "2026-08-20T23:38:54", "event": "start", "id": "TASK-141", "title": "a row stays blocked after its blockers close, because the stored status masks the computed one", "track": "intake", "actor": "agent", "from": "not_started", "to": "in_progress"} +{"ts": "2026-08-20T23:38:54", "event": "evidence", "id": "TASK-141", "title": "a row stays blocked after its blockers close, because the stored status masks the computed one", "track": "intake", "actor": "agent", "from": "—", "to": "evidence/2026-08/TASK-141-spec.md"} +{"ts": "2026-08-20T23:52:06", "event": "status", "id": "TASK-120", "title": "the linkage edges are read but never folded into KR progress", "track": "main", "actor": "agent", "depends_on": [], "from": "in_progress", "to": "review", "reason": ""} +{"ts": "2026-08-20T23:52:06", "event": "evidence", "id": "TASK-120", "title": "the linkage edges are read but never folded into KR progress", "track": "main", "actor": "agent", "from": "evidence/2026-08/TASK-120-spec.md", "to": "evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md"} +{"ts": "2026-08-20T23:52:24", "event": "add", "id": "TASK-144", "title": "the event log timestamp has no zone and the register has one, so ordering them is a guess", "track": "intake", "mode": "queue", "priority": "P1", "actor": "agent", "summary": "", "depends_on": [], "from": null, "to": "not_started"} +{"ts": "2026-08-20T23:52:24", "event": "add", "id": "TASK-145", "title": "the contract shape baseline is stale against its own recorder", "track": "intake", "mode": "queue", "priority": "P2", "actor": "agent", "summary": "", "depends_on": [], "from": null, "to": "not_started"} +{"ts": "2026-08-20T23:52:24", "event": "add", "id": "TASK-146", "title": "the viewer renders a KR current with no provenance because it does not go through the shared derivation", "track": "intake", "mode": "queue", "priority": "P2", "actor": "agent", "summary": "", "depends_on": [], "from": null, "to": "not_started"} +{"ts": "2026-08-20T23:53:36", "event": "start", "id": "TASK-143", "title": "two PRs each green on their own base merged into a red tree, and nothing checked the pair", "track": "intake", "actor": "agent", "from": "not_started", "to": "in_progress"} +{"ts": "2026-08-20T23:53:36", "event": "evidence", "id": "TASK-143", "title": "two PRs each green on their own base merged into a red tree, and nothing checked the pair", "track": "intake", "actor": "agent", "from": "—", "to": "evidence/2026-08/TASK-143-spec.md"} +{"ts": "2026-08-21T00:03:14", "event": "status", "id": "TASK-122", "title": "the repair path the tools advertise leaves the file needing a whitespace fix", "track": "main", "actor": "agent", "depends_on": [], "from": "in_progress", "to": "review", "reason": ""} +{"ts": "2026-08-21T00:03:14", "event": "evidence", "id": "TASK-122", "title": "the repair path the tools advertise leaves the file needing a whitespace fix", "track": "main", "actor": "agent", "from": "evidence/2026-08/TASK-122-spec.md", "to": "evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md"} +{"ts": "2026-08-21T00:03:52", "event": "add", "id": "TASK-147", "title": "nothing outside describe_cell proves the table and bullet paths stay separated", "track": "intake", "mode": "queue", "priority": "P2", "actor": "agent", "summary": "", "depends_on": [], "from": null, "to": "not_started"} diff --git a/perry/BOARD.md b/perry/BOARD.md index 276a2f58..0e49397e 100644 --- a/perry/BOARD.md +++ b/perry/BOARD.md @@ -36,18 +36,19 @@ | TASK-102 | Evidence becomes a typed relation: {path, kind, round}, not one prose cell | Coding Agent | not_started | — | — | V4 | TASK-090, TASK-092 | main | | | | | | | | TASK-114 | aiMark reads Perry through the current contracts instead of a pin nine versions old | Coding Agent | in_progress | delegated to an aiMark coding agent; awaiting paste-back | evidence/2026-08/TASK-114-delegation-prompt.md | V4 | — | main | | | | | | | | TASK-119 | the linkage graph is documented as machine-written and no tool writes it | Coding Agent | not_started | — | — | V3 | | main | | | | | | | -| TASK-120 | the linkage edges are read but never folded into KR progress | Coding Agent | in_progress | dispatched to claude-subagent; worktree pinned to 7c0bb99; state-schema.json scoped out so the gate passes without a release | evidence/2026-08/TASK-120-spec.md | V3 | — | main | | | | | | | +| TASK-120 | the linkage edges are read but never folded into KR progress | Coding Agent | review | PR #24 — contract 2.1; four findings handed back, two worth their own rows | evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md | V3 | — | main | | | | | | | | TASK-121 | the sweep that found four more live-state assertions runs once and then is thrown away | Coding Agent | not_started | — | — | V3 | | main | | | | | | | -| TASK-122 | the repair path the tools advertise leaves the file needing a whitespace fix | Coding Agent | in_progress | dispatched to claude-subagent; worktree pinned to 6c01b93; spec carries a live reproduction | evidence/2026-08/TASK-122-spec.md | V3 | — | main | | | | | | | +| TASK-122 | the repair path the tools advertise leaves the file needing a whitespace fix | Coding Agent | review | PR #25 — item 2 proved both ways on one run; two questions handed back | evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md | V3 | — | main | | | | | | | | TASK-123 | the goals writer takes the file as truth and derives the store, which is the opposite direction from the KR | Coding Agent | not_started | — | — | V4 | | main | | | | | | | | TASK-126 | closing the dangling-id row requires writing the record that re-dangles it | Coding Agent | review | PR #22 — the suite is fully green; verify the strong anti-vacuity case survives review, then close at V3 | evidence/2026-08/TASK-126-dispatch-2026-08-21-result.md | V3 | TASK-112 | main | | | | | | | | TASK-129 | Agent is five strings that do not join, and role has never once been written | Coding Agent | not_started | unblocked: work owns .perry/agents.jsonl → .perry/roles/ as of the 2026-08-20 signature; needs a spec, then dispatch | — | V3 | TASK-128 | main | | | | | | | | TASK-135 | a track can be declared but no existing row can be moved onto it | Coding Agent | not_started | — | — | V3 | | main | | | | | | | | TASK-136 | a queue track SLA is parsed, stored and never measured against anything | Coding Agent | not_started | — | — | V3 | | main | | | | | | | | TASK-140 | every mode contract slot is assigned to an axis, and the spine-to-unit map is written down | Coding Agent | review | PR #23 — three open questions for the user, incl. whether an empty illegal-pair list discharges § 7 risk 2 | evidence/2026-08/TASK-140-dispatch-2026-08-21-result.md | V3 | — | main | | | | | | | -| TASK-141 | a row stays blocked after its blockers close, because the stored status masks the computed one | Coding Agent | not_started | — | — | V3 | | intake | triaged | | 2026-08-20 | | | | +| TASK-141 | a row stays blocked after its blockers close, because the stored status masks the computed one | Coding Agent | in_progress | dispatched to claude-subagent; worktree pinned to f42e84b | evidence/2026-08/TASK-141-spec.md | V3 | — | intake | triaged | | 2026-08-20 | | | | | TASK-142 | triage has no check for a row stranded by a process bug, and the one signal that fired was read as prose hygiene | Coding Agent | not_started | design question answered 2026-08-20: it belongs in conformance, which triage already reads at step 0.5 — not as a new triage feature | — | V3 | — | intake | triaged | | 2026-08-20 | | | | -| TASK-143 | two PRs each green on their own base merged into a red tree, and nothing checked the pair | Coding Agent | not_started | — | — | V3 | | intake | triaged | | 2026-08-20 | | | | +| TASK-143 | two PRs each green on their own base merged into a red tree, and nothing checked the pair | Coding Agent | in_progress | dispatched to claude-subagent; worktree pinned to a10f897 | evidence/2026-08/TASK-143-spec.md | V3 | — | intake | triaged | | 2026-08-20 | | | | +| TASK-144 | the event log timestamp has no zone and the register has one, so ordering them is a guess | Coding Agent | not_started | — | — | V3 | | intake | triaged | | 2026-08-20 | | | | ## P2 @@ -68,6 +69,9 @@ | TASK-132 | the parity check cannot see 23 keys because Perry own state leaves four collections empty | Coding Agent | not_started | — | — | V3 | | main | | | | TASK-137 | a new queue row is born in the second stage, not the first | Coding Agent | not_started | — | — | V2 | | main | | | | TASK-139 | a design back-reference lives in a cell the close path clears, so a finished design reports as never handed off | Coding Agent | not_started | — | — | V3 | TASK-102 | intake | triaged | 2026-08-20 | +| TASK-145 | the contract shape baseline is stale against its own recorder | Coding Agent | not_started | — | — | V2 | | intake | triaged | 2026-08-20 | +| TASK-146 | the viewer renders a KR current with no provenance because it does not go through the shared derivation | Coding Agent | not_started | — | — | V3 | | intake | triaged | 2026-08-20 | +| TASK-147 | nothing outside describe_cell proves the table and bullet paths stay separated | Coding Agent | not_started | — | — | V3 | | intake | triaged | 2026-08-21 | ## Cadence (recurring; doesn't consume P0 slots) diff --git a/perry/evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md b/perry/evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md new file mode 100644 index 00000000..64a962a6 --- /dev/null +++ b/perry/evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md @@ -0,0 +1,106 @@ +# TASK-120 — result + +> Date: 2026-08-21 · Executor: claude-subagent · PR: https://github.com/ranjiao/Perry/pull/24 +> Branch: `coding/task-120-kr-progress-provenance` · Cycle time: ~35 min +> `perry-goals/list` **2.0 → 2.1**, additive: four keys added, none removed or +> retyped. `schema/state-schema.json` **untouched** — verified by diff, and the +> hard stop was live, so this is a real avoidance rather than luck. + +## The shape: Perry exposes the contradiction, it does not resolve it + +One derivation in `bin/lib/__init__.py`, three sibling keys per KR, emitted by +both payloads. Verified on this repository: + +``` +P-O1.1 current 0.0 target 1.0 + provenance state=asserted measured=false source=linkage-register + completion total 4 done 4 open 0 +P-O2.2 current 0.0 target 0.0 + provenance state=asserted measured=false + completion total 2 done 0 open 2 +``` + +`P-O1.1` no longer reads as 0-of-1 progress: it reads as **an author's assertion +of 0 against four closed tasks**, and Perry resolves neither. `P-O2.2` can no +longer be read as met, because nothing claims the zero was measured. + +**No `met` / `achieved` / `progress` / `ratio` key exists** — asserted as an +absence in the tests. No percentage is emitted anywhere: counts only, in their +own unit, so the tally cannot be misread as a metric value. That is the line the +spec drew and it held. + +## What it rejected, and one of the rejections is the interesting one + +- any ratio or percentage; +- a conformance entry for *"`current` disagrees with its tasks"* — that would be + Perry inferring the metric from the edges, which is the forbidden move one + step removed; +- **any new authored field in the register.** `asserted_at` reuses the + register's existing top-level `updated`, and `asserted_scope: "register"` is + emitted so a reader cannot mistake it for a per-KR date. **That choice is what + kept `schema/state-schema.json` out of the change** — the agent found the + cheap path around a gate rather than asking for a release. + +## The tally is what flips P-O1.1, not the staleness check + +Worth recording because it is counter-intuitive: P-O1.1 is **not stale by any +timestamp test** — all four of its tasks closed *before* the register's +`updated`. The contradiction is visible only because the completion tally sits +beside the number. A design that shipped staleness alone would have left that +KR reading exactly as wrongly as before. + +## Staleness, both directions on one fixture + +``` +not stale "no linked task has changed state since 2026-08-15T12:00:00" +stale "1 linked task changed state after 2026-08-15T12:00:00: + TASK-003 (in_progress → done)" + moved_tasks: [{id, from, to, at}] · only that KR goes stale +``` + +The fixture also carries a `next` event dated after the assertion, **so a check +keyed on the event name rather than on whether `to` is a status would redden.** + +## Parity, stated separately as instructed + +`perry-goals/list` **before**: 0 documented-not-emitted, **5** emitted-not-documented. +**After**: 0 and **5** — the same five, byte-identical, TASK-131's, not hidden +inside. Documented 54 → 77, emitted 59 → 78. Repo-wide total unchanged at 17. + +## Four findings handed back + +1. **The `Z` problem, unresolved and documented in the contract.** The register + writes `updated` as ISO with a `Z`; `.perry/events.jsonl` writes `ts` as naive + local time. There is no honest conversion, so the `Z` is **stripped rather + than applied**, and a register written within a few hours of a task move can + order wrongly. Fixing it means deciding what the event log's `ts` means — a + row of its own, touching every consumer. +2. **`tests/fixtures/contract-shapes.json` is stale w.r.t. its own recorder.** + `--record` wants to add an `empty_lists` block to two contracts and drop a + trailing newline. The agent refused to re-record rather than hide unrelated + drift in this row. **Someone will eventually re-record it into an unrelated + diff.** +3. **`viewer/serve.py` renders the chain view from `viewer/parsers.py`**, not + through `bin/lib`, so the viewer still shows `current` with no provenance. +4. **`measured` is `false` everywhere, by construction, until something re-runs + a metric.** Honest — and it means `stale` is Perry's only mechanical opinion + about a KR number. + +## Two notes on this project's own rules + +- The `current: 0` default the spec targeted lives in + `goals/state/linkage_TEMPLATE.md:13`, which writes it into every new register. + That is **authoring**, belongs to TASK-119's writer, and was correctly not + touched. `_num()` already returned `None` for an unwritten value, so V3 item 2 + was true but untested; it is now locked by four tests. +- **TASK-091's Definition of Done makes `bin/` a place where the history of that + defect cannot be written down.** The agent's first draft of a comment named the + deleted symbol and reddened `test_goals_writer`; it rewrote the comment to + describe the symbol without naming it. Same family as TASK-126 and TASK-112 — + a guard that forbids describing the thing it guards. + +## Process error, mine + +The worktree was cut from `feat/work-modes` **before** I committed the spec, so +the agent had to fetch it with `git checkout e71d7c0 -- `. Corrected for +TASK-122 and TASK-141: commit the spec, *then* cut the worktree. diff --git a/perry/evidence/2026-08/TASK-121-spec.md b/perry/evidence/2026-08/TASK-121-spec.md new file mode 100644 index 00000000..df9c78fe --- /dev/null +++ b/perry/evidence/2026-08/TASK-121-spec.md @@ -0,0 +1,85 @@ +# TASK-121 — the sweep that finds checks reading live state runs once and is thrown away + +> Source: `perry/evidence/2026-08/TASK-113-dispatch-2026-08-20-1813.md` +> Dispatch mode: auto +> Executor: claude-subagent +> Estimated cycle: medium +> Subjective verification: no +> Touches architecture: no — it adds a guard over the test suite +> Deployed: no + +## Schema + +- **Owner**: Coding Agent +- **Priority**: P1 +- **Attribution**: unlinked + +## The class, and its eight known instances + +A check that reads **the project living around it** as its expected value goes +red on ordinary progress, and green again for reasons that have nothing to do +with what it measures. Every instance below is real and dated within four days: + +| # | check | what moved under it | +|---|---|---| +| 1–3 | `test_diagnose`, `test_one_line_break_rule`, `test_v5_signoff` | TASK-113 found and fixed three | +| 4 | a fourth, handed over mid-run | same row | +| 5 | a fifth the agent found itself — `DESIGN-900` | same row | +| 6 | `test_md_store § test_config_including_its_prose_section` | asserted every config record is a `setting`; **declaring one track reddened it** | +| 7 | `test_track_attribution § TestPerrysOwnProjectIsUnmoved` | asserted Perry itself has no track register; same declaration reddened it | +| 8 | `test_state_cost` ×2 | asserted `perry/tasks.jsonl` is unclaimed and `.perry/events.jsonl` rolls up under `.perry/` — **both true until PR #14 declared the two store files owned** | + +TASK-113 fixed instances 1–5 **by hand, in one pass, and the pass was thrown +away.** Instances 6–8 arrived afterwards. There is no mechanism; there is a +memory of having looked. + +## Deliverable + +A guard that finds this class **mechanically**, so the next instance is reported +rather than discovered by a human running the suite after a merge. + +**What "this class" is, precisely, is the hard part of this row** — and getting +it wrong in either direction makes the guard worthless: + +- too broad, and it flags every test that reads a fixture, which is all of them; +- too narrow, and it is a list of the eight above wearing a regex. + +The instances give you the shape to generalise from: each one asserted a +**literal about the project's current state** — a count, an id, a set membership, +a filename — where the *property* being tested was true independently of that +literal. Note that instance 8's literals were about **which paths the schema declares +Perry owns**, not about a board row — so a guard keyed only on `BOARD.md` or the +task store would have missed it. + +Report what you decided the class is, in the guard's own docstring, in the voice +of the surrounding modules — and **name what it deliberately does not catch.** + +## Verification — V3 + +1. **It finds instances it was not shown.** Reconstruct at least three of the + eight from git history — `test_md_store` and `test_track_attribution` before + their 2026-08-21 fixes, and one of TASK-113's — and show the guard flags them. + Reconstruct, do not hand-write an approximation. +2. **It does not flag the fixes.** The same three, after their repairs, are + clean. A guard that still flags the repaired form is measuring the wrong + thing. +3. **False-positive floor, stated as a number.** Run it over the whole suite as + it stands and report **every** hit. If the count is not zero, each survivor + is either a real instance — open a row for it — or a false positive you must + name and explain. **Do not silence one to reach zero.** +4. **Reverting the guard reddens its own test.** +5. `python3 tests/parallel -j 4`, `bash tests/run`, `python3 bin/perry-lint`, + `git diff --check`. + +## Files in scope + +- the guard, as a new test module or a check under `tests/` +- its own tests and fixtures + +## Out of scope + +- **Fixing any instance you find.** Report them; each is its own row. This row + ships the mechanism, not the repairs. +- `bin/perry-diagnose` and `tests/test_diagnose.py` — an unmerged branch (PR #22) + is editing both. Cutting across it would conflict. +- `perry/` — no project state changes; `git diff -- perry/` must end empty. diff --git a/perry/evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md b/perry/evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md new file mode 100644 index 00000000..602635e2 --- /dev/null +++ b/perry/evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md @@ -0,0 +1,105 @@ +# TASK-122 — result + +> Date: 2026-08-21 · Executor: claude-subagent · PR: https://github.com/ranjiao/Perry/pull/25 +> Branch: `coding/task-122-bullet-padding` · Cycle time: ~50 min +> Code diff **2 files, +129/−5** — `bin/perry_store.py` +24/−5, +> `tests/test_md_store.py` +105/−0. (PR reports 32 files: the unpushed-ancestor +> sweep, fourth occurrence.) + +## The rule, and where the why is written + +> **"A table cell has boundaries; a bullet slot has neighbours"** — so only a +> cell may be handed padding it did not come with. + +`render_line` joins cells on `|`, which carries no whitespace of its own, so a +cell arriving as `single` must leave as `| split |`. A bullet slot is joined on +`""` between literal spans that **already hold every character around it** — the +span before the slot in `- Repo layout: single` is `'- Repo layout: '`, +separator space included. + +Mechanically it is **one variable**: `pad = " " if escape else ""`, used at both +places that invented padding — the disagreement branch (the reproduction) and the +whitespace-only branch. No new function, no branch keyed on a field name; the +same `escape` seam that already decided pipe-escaping now decides one more thing. + +It also **corrected a docstring that had gone false**: `cell_text` claimed +escaping was *"the ONE thing that differs between a table cell and a bullet +slot"*. Leaving that would be the stale prose that lets the next reader +re-introduce the padding. + +## Item 2, both ways, on one run + +Reverting **only** the rule, tests untouched: + +| reddened (3, all bullet) | did **not** redden (5, incl. the whole cell side) | +|---|---| +| `..._renders_byte_exact` | `test_a_table_cell_that_lost_its_padding_is_still_given_it_back` | +| `..._ends_without_a_trailing_space` | `test_a_track_row` (a board cell that disagrees) | +| `test_the_advertised_repair_survives_git_diff_check` | `test_a_config_setting`, both blank-marker tests | + +The sharpest case-2 test is the first in the right column: it is the +**padding-invention path itself**, `single` → `| split |`. It stayed green. +**The two paths are separated by `escape`; there is no bigger finding here.** + +The revert's own output is the bug verbatim: + +``` +AssertionError: Tuples differ: + (2, '.perry/config.md:6: trailing whitespace.\n+- State root: perry \n', '') + != (0, '', '') +``` + +## Item 4, run end to end rather than asserted + +A real `.perry/config.md` copied to a temp git repo, **declared through +`perry-conform declare`** — an earlier attempt with a bogus gate line was +correctly refused by the ADR-004 gate and redone — store written, drift planted, +then the advertised repair: + +``` +perry-config: rendered .../.perry/config.md from 9 stored record(s) +git diff --check exit=0 (silent) +6:- State root: perry$ ← cat -et; the $ is end-of-line, no trailing space +``` + +It is now a test, **and it asserts the value was actually restored** — otherwise +a clean `--check` would just be the cleanliness of a file nothing happened to. + +## One deliberate change beyond the reproduction + +A bullet whose slot is whitespace-only and whose store gains a value: +`- Code repo path: ` → before `- Code repo path: value ` (trailing space), after +`- Code repo path: value`. It kept the input's own space (`pad or raw`) rather +than dropping it, so the line neither gains whitespace nor loses what the author +wrote. **No file in the repo exercises this** — Perry's config uses `—` for +empties — but leaving that branch on the old rule would have made the fix a patch +instead of a rule. + +## It corrected the PMO, and the PMO was wrong + +The dispatch prompt said `test_diagnose`'s red *"has since been fixed on a +sibling branch"*. **It has not.** TASK-126's fix is on PR #22, **unmerged**, so +`feat/work-modes` still carries the red and every worktree cut from it inherits +it. The agent measured its own baseline, found the red, and **reported it rather +than absorbing it** — which is the behaviour every dispatch prompt asks for, +applied to the prompt itself. + +Worse, confirmed afterwards: the list is now +`['DESIGN-900', 'REL-00', 'ZZZ-404']`. **`ZZZ-404` came from +`TASK-126-spec.md` and `TASK-126-dispatch-2026-08-21-result.md`, both written by +the PMO** — the anti-vacuity example quoted in them. Writing the record about the +self-reference defect added a third instance of it. Third occurrence today; +first one caused while documenting the fix. + +## Two questions handed back + +1. A bullet slot genuinely empty in the source — `- Code repo path:` with no + space — renders `- Code repo path:value`. Faithful to *no whitespace the input + did not have*, and ugly. Unreachable from any current file; if a + `- Label: value` shape is wanted, that is a **normalization** rule and belongs + with whoever owns the config's shape, not in a renderer whose contract is + byte-comparison. +2. Nothing outside `describe_cell` proves the two paths stay separated. If a + third caller ever passes `escape=False` for a reason other than *"not in a + table"*, the flag's two meanings come apart — worth a conformance-style check + that the only producers of `escape=False` are slot descriptors. diff --git a/perry/evidence/2026-08/TASK-143-spec.md b/perry/evidence/2026-08/TASK-143-spec.md new file mode 100644 index 00000000..e2cff986 --- /dev/null +++ b/perry/evidence/2026-08/TASK-143-spec.md @@ -0,0 +1,85 @@ +# TASK-143 — two PRs each green on their own base merged into a red tree + +> Source: `.github/workflows/ci.yml`, and the merge that proved it +> Dispatch mode: auto +> Executor: claude-subagent +> Estimated cycle: small +> Subjective verification: no +> Touches architecture: no — one CI job's checkout, plus whatever it takes to +> report which pair disagreed +> Deployed: no + +## Schema + +- **Owner**: Coding Agent +- **Priority**: P1 +- **Attribution**: unlinked + +## Measured 2026-08-21, on the merge that had just happened + +`PR #14` (TASK-100) put both store files into `claims[]` at `e3f8621`. +`PR #15` (TASK-110) shipped `tests/test_state_cost.py` asserting the +**pre-claim** world — `perry/tasks.jsonl (unclaimed)` present, and +`.perry/events.jsonl` rolling up under the `.perry/` row. + +Each PR was **green on its own base.** The merged tree had **two red tests +neither PR could have seen**, and nobody found out until a human ran the suite +after the fact. + +The workflow is: + +```yaml +.github/workflows/ci.yml +on: + push: { branches: [main] } + pull_request: +``` + +A `pull_request` event checks out the **merge result** by default in GitHub +Actions — so the shape of the fix is not necessarily "check out something +different". **Establish what this workflow actually tested for #14 and #15 +before changing anything**; if the merge result was already what ran, the defect +is that each PR was tested against a base that then moved, and the fix is a +re-check at merge time rather than a different checkout. + +**Do not assume the diagnosis. Reproduce it.** + +## Deliverable + +A merge into the integration branch is checked against the **merged result**, +not only against each PR's own base — and when a pair disagrees, the report says +**which pair**, not just that something is red. + +Whether that is a workflow change, a job that re-runs on the branch tip, or a +required check that re-evaluates when the base moves, is yours to determine from +what you find. State the mechanism you rejected and why. + +## Verification — V3 + +1. **Reconstruct the pair.** From this repository's history, build the two + commits — the `claims[]` addition and the pre-claim test — and show your + mechanism **reports red before the merge lands**, where the old one did not. + This is the whole row; a change that cannot reproduce the original miss has + not been shown to fix it. +2. **A pair that genuinely does not interact still passes.** Two independent + changes must not be reported as conflicting. Without this the mechanism is + "always re-run everything and hope", which is not a check. +3. **The report names the pair.** Not "the suite is red" — which PR's change, + against which other, produced it. If the mechanism cannot attribute, say so + plainly rather than shipping a signal nobody can act on. +4. `python3 tests/parallel -j 4`, `python3 bin/perry-lint`, `git diff --check`. + +## Files in scope + +- `.github/workflows/ci.yml` +- a helper script under `tests/` if the mechanism needs one +- documentation of the mechanism where a contributor will meet it + +## Out of scope + +- **`perry/` — no project state changes.** `git diff -- perry/` must end empty. +- Fixing the two `test_state_cost` assertions. They were repaired on + `feat/work-modes` at `13cfe2f`; you are preventing the class, not that instance. +- Any change to what the suite runs, or to `tests/parallel` and `tests/run`. +- Branch protection settings and anything requiring repository admin — if the + honest fix needs one, **say so and stop**; that is the user's to apply. diff --git a/perry/journal/2026-08/2026-08-20.md b/perry/journal/2026-08/2026-08-20.md index c4a97a5b..f7c43a8c 100644 --- a/perry/journal/2026-08/2026-08-20.md +++ b/perry/journal/2026-08/2026-08-20.md @@ -179,6 +179,15 @@ - [TASK-122] evidence · — → evidence/2026-08/TASK-122-spec.md - [TASK-140] in_progress → review - [TASK-140] evidence · evidence/2026-08/TASK-140-dispatch-2026-08-21.md → evidence/2026-08/TASK-140-dispatch-2026-08-21-result.md +- [TASK-141] not_started → in_progress · started +- [TASK-141] evidence · — → evidence/2026-08/TASK-141-spec.md +- [TASK-120] in_progress → review +- [TASK-120] evidence · evidence/2026-08/TASK-120-spec.md → evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md +- [TASK-144] — → not_started · the event log timestamp has no zone and the register has one, so ordering them is a guess · owner: Coding Agent · priority: P1 +- [TASK-145] — → not_started · the contract shape baseline is stale against its own recorder · owner: Coding Agent · priority: P2 +- [TASK-146] — → not_started · the viewer renders a KR current with no provenance because it does not go through the shared derivation · owner: Coding Agent · priority: P2 +- [TASK-143] not_started → in_progress · started +- [TASK-143] evidence · — → evidence/2026-08/TASK-143-spec.md ## Notes @@ -763,6 +772,39 @@ - **Out of scope**: — - **KR linkage**: unlinked +### TASK-144 — the event log timestamp has no zone and the register has one, so ordering them is a guess + +- **Owner**: Coding Agent +- **Priority**: P1 +- **Track / mode**: intake / queue +- **Deliverable**: what .perry/events.jsonl ts means is decided and written down — naive local, UTC, or offset-carrying — and every producer and consumer agrees, so a register updated field and an event can be ordered without stripping a zone +- **Verification**: measured 2026-08-21 by TASK-120: the register writes updated as ISO with a Z and the event log writes ts as naive local with none, so the Z is stripped rather than applied and a register written within hours of a task move can order wrongly. After: a fixture whose register assertion and task move straddle a zone boundary orders correctly, and reverting reddens it +- **Dependencies**: — +- **Out of scope**: — +- **KR linkage**: unlinked + +### TASK-145 — the contract shape baseline is stale against its own recorder + +- **Owner**: Coding Agent +- **Priority**: P2 +- **Track / mode**: intake / queue +- **Deliverable**: tests/fixtures/contract-shapes.json matches what --record produces, so re-recording it is a no-op and no unrelated drift can ride into a future diff +- **Verification**: measured 2026-08-21 by TASK-120: --record wants to add an empty_lists block to perry-decide/list and perry-task/list and to drop the file trailing newline. After: --record produces a byte-identical file; introducing a real shape change still makes it differ +- **Dependencies**: — +- **Out of scope**: — +- **KR linkage**: unlinked + +### TASK-146 — the viewer renders a KR current with no provenance because it does not go through the shared derivation + +- **Owner**: Coding Agent +- **Priority**: P2 +- **Track / mode**: intake / queue +- **Deliverable**: viewer/serve.py chain view shows a KR current together with whether it was asserted or measured and whether it has gone stale, by reading perry-state --json or the shared bin/lib derivation rather than re-deriving from viewer/parsers.py +- **Verification**: measured 2026-08-21 by TASK-120: the viewer reads viewer/parsers.py and never touches bin/lib, so P-O1.1 renders as 0-of-1 with no sign that the 0 is an assertion contradicted by four closed tasks. After: the same KR renders its provenance, and a KR whose assertion is stale is visibly marked +- **Dependencies**: — +- **Out of scope**: — +- **KR linkage**: unlinked + ## V5 sign-off **TASK-107 — V5 sign-off. Ran Jiao, 2026-08-20.** diff --git a/perry/journal/2026-08/2026-08-21.md b/perry/journal/2026-08/2026-08-21.md new file mode 100644 index 00000000..f43110ad --- /dev/null +++ b/perry/journal/2026-08/2026-08-21.md @@ -0,0 +1,20 @@ +# 2026-08-21 + +## Status changes + +- [TASK-122] in_progress → review +- [TASK-122] evidence · evidence/2026-08/TASK-122-spec.md → evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md +- [TASK-147] — → not_started · nothing outside describe_cell proves the table and bullet paths stay separated · owner: Coding Agent · priority: P2 + +## New tasks added + +### TASK-147 — nothing outside describe_cell proves the table and bullet paths stay separated + +- **Owner**: Coding Agent +- **Priority**: P2 +- **Track / mode**: intake / queue +- **Deliverable**: a conformance-style check that the only producers of escape=False are slot descriptors, so the flag cannot quietly acquire a second meaning: today it answers both is this inside a table and may this be handed padding it did not come with, and a third caller passing it for some other reason splits those two apart with nothing to notice +- **Verification**: measured 2026-08-21 by TASK-122: the new padding class proves the separation for padding only, and no check covers the flag itself. After: introducing a caller that passes escape=False from outside a slot descriptor reddens, and the existing legitimate producers do not +- **Dependencies**: — +- **Out of scope**: — +- **KR linkage**: unlinked diff --git a/perry/tasks.jsonl b/perry/tasks.jsonl index 086c1629..4479f8c3 100644 --- a/perry/tasks.jsonl +++ b/perry/tasks.jsonl @@ -127,14 +127,18 @@ {"id": "TASK-094", "title": "Delete the header rule and the row splitter for the three stores", "owner": "Coding Agent", "status": "review", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-094-dispatch-2026-08-20-1958.md", "next_action": "PR #20 merged but the row does NOT close on it: verification item 1 asked for 0 call sites and BOARD.md keeps 13 splits / 87 resolutions on four storeless registers — needs a scope decision, not a close", "depends_on": ["TASK-090", "TASK-092"], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-19T10:27:51", "order": 1, "summary": ""} {"id": "TASK-037", "title": "perry-goals writer", "owner": "Coding Agent", "status": "not_started", "priority": "P2", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V4", "evidence": "—", "next_action": "unblocked: TASK-092 closed 2026-08-20. Re-scope per its own note — flag naming and the module-scope handler defect only; the rest was overtaken by TASK-092 and TASK-123", "depends_on": ["TASK-092"], "commitment": "", "parent": "", "group": "P2", "role": "", "created": "2026-08-17T00:55:19", "order": 0, "summary": ""} {"id": "TASK-045", "title": "Retire the runtime tolerance branches, behind the conformance marker", "owner": "Coding Agent", "status": "not_started", "priority": "P2", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V4", "evidence": "—", "next_action": "unblocked: the whole chain closed — TASK-044 and TASK-047 are both done. The conformance marker enforces on this branch, which is the precondition this row was waiting for", "depends_on": ["TASK-044", "TASK-047"], "commitment": "", "parent": "", "group": "P2", "role": "", "created": "2026-08-17T15:59:34", "order": 2, "summary": ""} -{"id": "TASK-141", "title": "a row stays blocked after its blockers close, because the stored status masks the computed one", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P1", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "—", "next_action": "—", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T21:21:09", "order": 17} {"id": "TASK-142", "title": "triage has no check for a row stranded by a process bug, and the one signal that fired was read as prose hygiene", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P1", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "—", "next_action": "design question answered 2026-08-20: it belongs in conformance, which triage already reads at step 0.5 — not as a new triage feature", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T21:24:30", "order": 18} -{"id": "TASK-143", "title": "two PRs each green on their own base merged into a red tree, and nothing checked the pair", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P1", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "—", "next_action": "—", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T23:09:32", "order": 19} {"id": "TASK-100", "title": "tasks.jsonl is in no claims[] entry, so a namespace collision on it cannot be reported", "owner": "Coding Agent", "status": "done", "priority": "P2", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-100-dispatch-2026-08-20-1730.md", "next_action": "PR merged; ready to close once origin is merged down and the suite re-run locally", "depends_on": [], "commitment": "", "parent": "", "group": "P2", "role": "", "created": "2026-08-19T11:44:02", "order": null, "summary": ""} {"id": "TASK-111", "title": "a test reads two files outside the repository, so it is green here and red on CI forever", "summary": "", "owner": "Coding Agent", "status": "done", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-111-dispatch-2026-08-20-1930.md", "next_action": "PR merged; ready to close once origin is merged down and the suite re-run locally", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T16:38:31", "order": null} {"id": "TASK-127", "title": "the contract docs and the payloads they describe are never diffed against each other", "summary": "", "owner": "Coding Agent", "status": "done", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-127-dispatch-2026-08-20-2045.md", "next_action": "PR merged; ready to close once origin is merged down and the suite re-run locally", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T19:57:28", "order": null} {"id": "TASK-133", "title": "declare the first non-project track on Perry itself, and measure what a mixed spine costs", "summary": "", "owner": "User + Agent", "status": "done", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-133-track-experiment.md", "next_action": "PR merged; ready to close once origin is merged down and the suite re-run locally", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T20:29:08", "order": null} -{"id": "TASK-120", "title": "the linkage edges are read but never folded into KR progress", "summary": "", "owner": "Coding Agent", "status": "in_progress", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-120-spec.md", "next_action": "dispatched to claude-subagent; worktree pinned to 7c0bb99; state-schema.json scoped out so the gate passes without a release", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T17:58:10", "order": 8} {"id": "TASK-126", "title": "closing the dangling-id row requires writing the record that re-dangles it", "summary": "", "owner": "Coding Agent", "status": "review", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-126-dispatch-2026-08-21-result.md", "next_action": "PR #22 — the suite is fully green; verify the strong anti-vacuity case survives review, then close at V3", "depends_on": ["TASK-112"], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T19:57:10", "order": 12} -{"id": "TASK-122", "title": "the repair path the tools advertise leaves the file needing a whitespace fix", "summary": "", "owner": "Coding Agent", "status": "in_progress", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-122-spec.md", "next_action": "dispatched to claude-subagent; worktree pinned to 6c01b93; spec carries a live reproduction", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T18:30:44", "order": 10} {"id": "TASK-140", "title": "every mode contract slot is assigned to an axis, and the spine-to-unit map is written down", "summary": "", "owner": "Coding Agent", "status": "review", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-140-dispatch-2026-08-21-result.md", "next_action": "PR #23 — three open questions for the user, incl. whether an empty illegal-pair list discharges § 7 risk 2", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T21:08:52", "order": 16} +{"id": "TASK-141", "title": "a row stays blocked after its blockers close, because the stored status masks the computed one", "summary": "", "owner": "Coding Agent", "status": "in_progress", "priority": "P1", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "evidence/2026-08/TASK-141-spec.md", "next_action": "dispatched to claude-subagent; worktree pinned to f42e84b", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T21:21:09", "order": 17} +{"id": "TASK-120", "title": "the linkage edges are read but never folded into KR progress", "summary": "", "owner": "Coding Agent", "status": "review", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-120-dispatch-2026-08-21-result.md", "next_action": "PR #24 — contract 2.1; four findings handed back, two worth their own rows", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T17:58:10", "order": 8} +{"id": "TASK-144", "title": "the event log timestamp has no zone and the register has one, so ordering them is a guess", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P1", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "—", "next_action": "—", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T23:52:24", "order": 20} +{"id": "TASK-145", "title": "the contract shape baseline is stale against its own recorder", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P2", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V2", "evidence": "—", "next_action": "—", "depends_on": [], "commitment": "", "parent": "", "group": "P2", "role": "", "created": "2026-08-20T23:52:24", "order": 15} +{"id": "TASK-146", "title": "the viewer renders a KR current with no provenance because it does not go through the shared derivation", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P2", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "—", "next_action": "—", "depends_on": [], "commitment": "", "parent": "", "group": "P2", "role": "", "created": "2026-08-20T23:52:24", "order": 16} +{"id": "TASK-143", "title": "two PRs each green on their own base merged into a red tree, and nothing checked the pair", "summary": "", "owner": "Coding Agent", "status": "in_progress", "priority": "P1", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-20", "verification": "V3", "evidence": "evidence/2026-08/TASK-143-spec.md", "next_action": "dispatched to claude-subagent; worktree pinned to a10f897", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T23:09:32", "order": 19} +{"id": "TASK-122", "title": "the repair path the tools advertise leaves the file needing a whitespace fix", "summary": "", "owner": "Coding Agent", "status": "review", "priority": "P1", "track": "main", "stage": "", "stage_since": "", "arrived": "", "verification": "V3", "evidence": "evidence/2026-08/TASK-122-dispatch-2026-08-21-result.md", "next_action": "PR #25 — item 2 proved both ways on one run; two questions handed back", "depends_on": [], "commitment": "", "parent": "", "group": "P1", "role": "", "created": "2026-08-20T18:30:44", "order": 10} +{"id": "TASK-147", "title": "nothing outside describe_cell proves the table and bullet paths stay separated", "summary": "", "owner": "Coding Agent", "status": "not_started", "priority": "P2", "track": "intake", "stage": "triaged", "stage_since": "", "arrived": "2026-08-21", "verification": "V3", "evidence": "—", "next_action": "—", "depends_on": [], "commitment": "", "parent": "", "group": "P2", "role": "", "created": "2026-08-21T00:03:52", "order": 17} diff --git a/tests/fixtures/live-state-expectations.json b/tests/fixtures/live-state-expectations.json new file mode 100644 index 00000000..d8d49788 --- /dev/null +++ b/tests/fixtures/live-state-expectations.json @@ -0,0 +1,65 @@ +{ + "note": "Every hit of tests/live_state_expectations.py over this repository, with a verdict on each. Recorded rather than asserted empty because it is not empty: `instance` means the finding is real and owes a row of its own; `false positive` means the guard cannot see why it is fine and the reason is written down instead of the finding being silenced. Re-record with `python3 tests/live_state_expectations.py --record`, which keeps every verdict already written for a finding that is still there.", + "findings": [ + { + "module": "tests/test_contract_invariance.py", + "lineno": 208, + "test": "TestAnAdditionIsAllowedAndAnnounced.test_typed_status_alias_change_is_announced", + "assertion": "assertEqual", + "actual": "set(entry['fields'])", + "expected": "{'status_text', 'conformance.sections_read', 'conformance.sections_skipped', 'conformance.rows_\u2026", + "verdict": "false positive", + "why": "`semantics` is a changelog `bin/perry-task` carries in its own source. The payload reaches it through a `cwd=ROOT` run, and half one cannot tell a subtree the tool built out of its own constants from one it built out of the project. Editing the changelog SHOULD redden this test." + }, + { + "module": "tests/test_md_store.py", + "lineno": 127, + "test": "TestThisRepositoryIsReproducedByteForByte.test_okr", + "assertion": "assertGreater", + "actual": "len(krs)", + "expected": "20", + "verdict": "instance", + "why": "`assertGreater(len(krs), 20)` over `perry/OKR.md` \u2014 c9018ae's shape exactly, on the file rather than the board. Retiring five KRs would redden a test whose subject is byte-identical round-tripping. The independent count on the line above it (`len(krs) == kr_lines(text)`) is the property; the 20 is a snapshot standing beside it." + }, + { + "module": "tests/test_md_store.py", + "lineno": 300, + "test": "TestAMutatedStoreMovesTheFile.test_an_okr_kr_field", + "assertion": "assertEqual", + "actual": "drift[0]['store']", + "expected": "'2099-01-01'", + "verdict": "false positive", + "why": "`2099-01-01` is the value this test wrote into the store four lines earlier. It is the test's own input coming back, not a fact about the project, and it moves when the test moves." + }, + { + "module": "tests/test_md_store.py", + "lineno": 314, + "test": "TestAMutatedStoreMovesTheFile.test_a_config_setting", + "assertion": "assertEqual", + "actual": "[d['key'] for d in drift]", + "expected": "['setting/state_root']", + "verdict": "false positive", + "why": "`state_root` is the record this test selected and mutated, so the assertion moves with the test. The `setting/` half of the key IS read out of `.perry/config.md` \u2014 the coupling instance 6 died of \u2014 but it names a record the test chose rather than the whole register." + }, + { + "module": "tests/test_prioritize.py", + "lineno": 507, + "test": "TestEveryEventSaysWhatItsPairMeans.test_an_id_shaped_word_in_prose_is_warned_about", + "assertion": "assertEqual", + "actual": "fn('the ROUND-2 defect', ctx)", + "expected": "['ROUND-2']", + "verdict": "instance", + "why": "`ctx` is built from `load_task_records(perry/)`, this repository's live task store. The neighbouring line is the more fragile half and is NOT flagged, because `[]` is not a closed literal: `fn('see ADR-006 and USER-014', ctx) == []` requires both of those ids to still resolve on this board, and either one leaving reddens a test about prose parsing." + }, + { + "module": "tests/test_task_writer.py", + "lineno": 152, + "test": "TestFormatIsMechanized.test_every_hand_written_row_in_perrys_own_board_round_trips", + "assertion": "assertGreater", + "actual": "len(rows)", + "expected": "5", + "verdict": "instance", + "why": "`assertGreater(len(rows), 5)` over Perry's own `BOARD.md`. Same shape as c9018ae; a board that closes its way below six rows reddens a test whose subject is row formatting." + } + ] +} diff --git a/tests/fixtures/live-state/md_store.before.py b/tests/fixtures/live-state/md_store.before.py new file mode 100644 index 00000000..2cf7bb9d --- /dev/null +++ b/tests/fixtures/live-state/md_store.before.py @@ -0,0 +1,695 @@ +"""`OKR.md` and `.perry/config.md` as stores — TASK-092, ADR-007's second slice. + +**The bar is `cmp`, and every claim here is measured against it.** A store that +cannot reproduce the document it replaces has already lost data, and +"reproduce" has to mean the bytes: `TASK-037-spec` carries a manual verdict on +DESIGN-005 § 5.5's finding that "the failure mode is a file that still parses +and no longer reads the way its author wrote it". A byte comparison is exactly +the check that finding says does not exist — a file that no longer reads the +way its author wrote it fails one by definition. + +**Byte-identity alone is not evidence, and that is the point of half this +file.** A renderer that echoes the file back passes `cmp` on every project in +the world. So each round trip is guarded three ways: + + 1. the record count is compared against an INDEPENDENT count of the rows in + the file (`viewer/parsers.py`, and a regex over the raw lines); + 2. the report must show ZERO verbatim cells — every cell of every claimed + line came out of the store, not out of the file; + 3. a field is mutated in the store and the rendered file must MOVE with it. + A renderer that cannot be made to print a wrong value cannot be shown to + print a right one. + +**Two projects, because one project's file is a fixture wearing a disguise.** +`tests/fixtures/second-project/` is shaped on `~/proj/gimegime-pmo` — bullet +KRs rather than tables, Chinese prose, several version blocks, a config with +prose sections. `TestTheSecondRealProject` runs the same comparison against +that project itself when the machine has it, and skips when it does not; the +fixture is what holds the line everywhere else. + +Run: python3 tests/parallel test_md_store +""" + +from __future__ import annotations + +import json +import pathlib +import re +import shutil +import subprocess +import sys +import tempfile +import unittest + +ROOT = pathlib.Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "bin")) +sys.path.insert(0, str(ROOT / "viewer")) +import parsers as P # noqa: E402 +import perry_md_store as M # noqa: E402 + +from gate import GATE_OFF # noqa: E402 + +FIXTURES = ROOT / "tests" / "fixtures" +SECOND_PROJECT = pathlib.Path("~/proj/gimegime-pmo").expanduser() + +#: An independent count of the KR-bearing lines in a raw `OKR.md`, written +#: without importing anything the store uses. Two implementations of "how many +#: KRs are in this file" is the point here: if the scanner and this regex ever +#: agree only because they are the same code, the coverage assertion below +#: proves nothing. +KR_TABLE_ROW = re.compile(r"^\|\s*\**(?:KR|P)[-\w.]*\d\**\s*\|") +KR_BULLET = re.compile(r"^\s*-\s*\**(?:KR|P-O)[\w.\-]*\d\**[^::]*[::]") + + +def kr_lines(text: str) -> int: + return sum(1 for line in text.split("\n") + if KR_TABLE_ROW.match(line) or KR_BULLET.match(line)) + + +def run(tool: str, *args, root: pathlib.Path): + return subprocess.run( + [sys.executable, str(ROOT / "bin" / tool), *args, "--root", str(root)], + capture_output=True, text=True, cwd=str(ROOT)) + + +class RoundTrip: + """The three-way guard, in one place so no case can quietly skip a leg.""" + + def assert_round_trips(self, doc, path: pathlib.Path, *, + expect_kinds=None): + text = path.read_text(encoding="utf-8") + records = M.derive(doc, text) + rendered, report = M.render(doc, text, records) + + # 1. bytes. + self.assertEqual( + rendered, text, + f"{path} is not reproduced byte-identically; first difference " + f"{json.dumps(_first_difference(text, rendered), ensure_ascii=False)}") + + # 2. nothing was reproduced by echoing it back. + self.assertEqual( + report["cells_verbatim"], {}, + f"{path}: cells came out of the FILE rather than the store — " + f"byte-identity that proves nothing about the store") + self.assertEqual(report["lines_verbatim"], [], str(path)) + self.assertEqual(report["records_not_in_the_file"], [], str(path)) + self.assertEqual( + report["cells_wearing_decoration"], {}, + f"{path}: cells came back byte-identical by keeping text around " + f"the stored value — the other way `cmp` can pass on nothing") + + if expect_kinds is not None: + self.assertEqual(report["kinds"], expect_kinds, str(path)) + return records, report + + +def _first_difference(a_text: str, b_text: str) -> dict: + a, b = a_text.split("\n"), b_text.split("\n") + n = next((i for i in range(max(len(a), len(b))) + if (a[i:i + 1] or [None]) != (b[i:i + 1] or [None])), 0) + return {"line": n + 1, + "file": (a[n] if n < len(a) else "")[:200], + "rendered": (b[n] if n < len(b) else "")[:200]} + + +class TestThisRepositoryIsReproducedByteForByte(unittest.TestCase, RoundTrip): + """V4 step 1 and 2 — Perry's own two files, not a fixture.""" + + def test_okr(self): + records, _ = self.assert_round_trips(M.OKR, ROOT / "perry" / "OKR.md") + text = (ROOT / "perry" / "OKR.md").read_text() + krs = [r for r in records if r["kind"] == "kr"] + self.assertEqual( + len(krs), kr_lines(text), + "the store holds a different number of KRs than the file has KR " + "lines — a byte-identical render that dropped rows") + self.assertGreater(len(krs), 20, "a file this size holding <20 KRs " + "means the scanner stopped early") + + def test_config_including_its_prose_section(self): + path = ROOT / ".perry" / "config.md" + records, report = self.assert_round_trips(M.CONFIG, path) + # The section V4 step 2 names. It is PROSE: the store must hold no + # record for it and the renderer must not touch a byte of it. + self.assertIn("## Why the state root is not `.`", path.read_text()) + self.assertEqual(report["kinds"], {"setting": len(records)}) + keys = {r["key"] for r in records} + for expected in ("document_language", "state_root", "code_repo_path"): + self.assertIn(expected, keys) + + def test_the_declared_blank_marker_survives_the_bullet_path(self): + """`- Code repo path: —` — c9018ae's rule, on a line that is not a table. + + The marker is LAYOUT: it stays while the store's field is empty. If the + bullet path had grown its own blank rule, `—` would mean one thing in a + board cell and another in this file, which is exactly the second cell + model ADR-007 exists to remove. + """ + path = ROOT / ".perry" / "config.md" + text = path.read_text() + self.assertIn("- Code repo path: —", text) + records = M.derive(M.CONFIG, text) + rec = next(r for r in records if r["key"] == "code_repo_path") + self.assertEqual(rec["value"], "", + "a declared blank marker was stored as data") + self.assertEqual(M.render(M.CONFIG, text, records)[0], text) + + +class TestTheSecondProjectFixture(unittest.TestCase, RoundTrip): + """V4 step 3, in the form that runs everywhere. + + Shaped on `~/proj/gimegime-pmo`: bullet KRs instead of tables, several + version blocks, Chinese prose, a config carrying a `## Tracks` table and + screens of dispatch notes. **Neither real project on this machine declares + a `## Tracks` table**, which is the register `DESIGN-003 § 5.2` defines and + `KR-O1.3` is about — so the only place it can be held to `cmp` is here. + """ + + def test_okr_with_bullet_krs_and_a_commitments_register(self): + path = FIXTURES / "second-project" / "OKR.md" + records, report = self.assert_round_trips( + M.OKR, path, expect_kinds={"kr": 7, "commitment": 2, "version": 2}) + self.assertEqual(len([r for r in records if r["kind"] == "kr"]), + kr_lines(path.read_text())) + # Every KR here came from the bullet form, which is the half of + # `_parse_krs` a table-only store would have dropped entirely. + self.assertTrue(all(r["form"] == "bullet" + for r in records if r["kind"] == "kr")) + + def test_config_with_a_tracks_table_and_prose_sections(self): + path = FIXTURES / "second-project" / ".perry" / "config.md" + records, report = self.assert_round_trips( + M.CONFIG, path, expect_kinds={"setting": 8, "track": 3}) + tracks = [r for r in records if r["kind"] == "track"] + self.assertEqual([t["track"] for t in tracks], + ["main", "research", "ops"]) + self.assertEqual([t["mode"] for t in tracks], + ["project", "pipeline", "queue"]) + # `bin/perry-state § parse_tracks` is the shipped reader of this table. + # The store must hold what that reader reads, or the two have come + # apart on the register every non-`project` mode depends on. + declared = _parse_tracks(path.read_text()) + self.assertEqual([t["track"] for t in declared], + [t["track"] for t in tracks]) + # Compared through `stored_value`, because the two answer slightly + # different questions and the difference is the design: the shipped + # reader is TOLERANT and hands back the `—` a project wrote, while the + # store holds the typed value that marker stands for — empty. Asserting + # the raw cells matched would be asserting the store failed to + # normalise anything. + self.assertEqual([M.stored_value(t["sla"]) for t in declared], + [t["sla"] for t in tracks]) + self.assertEqual([M.stored_value(t["stages"]) for t in declared], + [t["stages"] for t in tracks]) + + def test_the_other_bundled_projects_round_trip_too(self): + for rel in ("sample-project/OKR.md", "sample-project-zh/OKR.md"): + with self.subTest(rel): + self.assert_round_trips(M.OKR, FIXTURES / rel) + self.assert_round_trips( + M.CONFIG, FIXTURES / "sample-project-zh" / ".perry" / "config.md") + + +def _parse_tracks(text: str) -> list[dict]: + """`bin/perry-state § parse_tracks`, loaded as the module it lives in.""" + import importlib.machinery + import importlib.util + path = ROOT / "bin" / "perry-state" + spec = importlib.util.spec_from_loader( + "perry_state_mod", + importlib.machinery.SourceFileLoader("perry_state_mod", str(path))) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod.parse_tracks(text) + + +@unittest.skipUnless( + (SECOND_PROJECT / "OKR.md").is_file(), + f"{SECOND_PROJECT} is not on this machine; " + f"tests/fixtures/second-project carries its shape") +class TestTheSecondRealProject(unittest.TestCase, RoundTrip): + """V4 step 3 against the project itself — on a COPY, never the original. + + It is the untidy one on purpose: a year of history, a board organized by + workstream, 61 lint errors, and a `Status: 半解` cell migration refuses to + coerce. Nothing here writes into it; the copy is what is read. + """ + + def copy(self) -> pathlib.Path: + d = pathlib.Path(tempfile.mkdtemp(prefix="perry-second-project-")) + self.addCleanup(shutil.rmtree, d, ignore_errors=True) + for rel in ("OKR.md", ".perry/config.md"): + src = SECOND_PROJECT / rel + if src.is_file(): + dst = d / rel + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(src, dst) + return d + + def test_okr_and_config_are_reproduced_byte_for_byte(self): + d = self.copy() + records, _ = self.assert_round_trips(M.OKR, d / "OKR.md") + self.assertEqual(len([r for r in records if r["kind"] == "kr"]), + kr_lines((d / "OKR.md").read_text())) + cfg = d / ".perry" / "config.md" + if cfg.is_file(): + self.assert_round_trips(M.CONFIG, cfg) + + +class TestAMutatedStoreMovesTheFile(unittest.TestCase): + """V4 step 4 — the guard that byte-identity is not an echo. + + Change one field in the store; the rendered file must change at exactly + that cell, and the drift report must NAME the cell. `describe_cell`'s own + docstring records the first version of this getting it wrong by falling + back to verbatim when the two disagreed, which meant the layout was being + derived against the store it was meant to be testing. + """ + + def test_an_okr_kr_field(self): + path = ROOT / "perry" / "OKR.md" + text = path.read_text() + records = M.derive(M.OKR, text) + target = next(r for r in records if r["kind"] == "kr" and r["deadline"]) + before = target["deadline"] + target["deadline"] = "2099-01-01" + + rendered, report = M.render(M.OKR, text, records) + self.assertNotEqual(rendered, text, "the store moved and the render " + "did not — the file is echoing") + self.assertIn("2099-01-01", rendered) + drift = report["cells_the_store_and_the_file_disagree_on"] + self.assertEqual(len(drift), 1, drift) + self.assertEqual(drift[0]["store"], "2099-01-01") + self.assertEqual(drift[0]["file"], before) + self.assertIn(target["id"], drift[0]["key"]) + + def test_a_config_setting(self): + path = ROOT / ".perry" / "config.md" + text = path.read_text() + records = M.derive(M.CONFIG, text) + target = next(r for r in records if r["key"] == "state_root") + target["value"] = "docs" + + rendered, report = M.render(M.CONFIG, text, records) + self.assertIn("- State root: docs", rendered) + drift = report["cells_the_store_and_the_file_disagree_on"] + self.assertEqual([d["key"] for d in drift], ["setting/state_root"]) + + def test_a_track_row(self): + path = FIXTURES / "second-project" / ".perry" / "config.md" + text = path.read_text() + records = M.derive(M.CONFIG, text) + target = next(r for r in records if r.get("track") == "ops") + target["sla"] = "7d" + + rendered, report = M.render(M.CONFIG, text, records) + self.assertIn("| 7d |", rendered) + drift = report["cells_the_store_and_the_file_disagree_on"] + self.assertEqual([(d["key"], d["column"], d["file"], d["store"]) + for d in drift], [("track/ops", "SLA", "3d", "7d")]) + + def test_a_blank_marker_is_replaced_once_the_store_has_a_value(self): + """The other direction of c9018ae's rule, which nothing else covers. + + `—` stays while the field is empty; the moment the store carries a + value, the marker is what gets replaced. A renderer that kept the + marker unconditionally would pass every test above. + """ + path = ROOT / ".perry" / "config.md" + text = path.read_text() + records = M.derive(M.CONFIG, text) + target = next(r for r in records if r["key"] == "code_repo_path") + self.assertEqual(target["value"], "") + self.assertIn("- Code repo path: —", M.render(M.CONFIG, text, records)[0]) + target["value"] = "/tmp/elsewhere" + rendered, _ = M.render(M.CONFIG, text, records) + self.assertIn("- Code repo path: /tmp/elsewhere", rendered) + self.assertNotIn("- Code repo path: —", rendered) + + +class Project: + """A throwaway project carrying Perry's own two files.""" + + def __init__(self, case: unittest.TestCase): + self.root = pathlib.Path(tempfile.mkdtemp(prefix="perry-md-store-")) + case.addCleanup(shutil.rmtree, self.root, ignore_errors=True) + (self.root / "perry").mkdir() + (self.root / ".perry").mkdir() + shutil.copy2(ROOT / "perry" / "OKR.md", self.root / "perry" / "OKR.md") + # The conformance gate reads `.perry/config.md` to decide its own mode, + # and `.perry/config.md` is one of the two files under test — so a + # fixture here is writing the very file the gate consults about + # itself. `GATE_OFF` is the documented way out (tests/gate.py); the + # gate's own branches are `tests/test_conformance.py`'s subject. + (self.root / ".perry" / "config.md").write_text( + (ROOT / ".perry" / "config.md").read_text() + GATE_OFF, + encoding="utf-8") + + def okr(self, *args): + return run("perry-okr", *args, root=self.root) + + def config(self, *args): + return run("perry-config", *args, root=self.root) + + def okr_text(self) -> str: + return (self.root / "perry" / "OKR.md").read_text() + + def config_text(self) -> str: + return (self.root / ".perry" / "config.md").read_text() + + +class TestTheCommandLine(unittest.TestCase): + def test_render_and_diff_refuse_before_a_store_exists(self): + """"Nothing to verify" rather than a pass. + + Rendering a file from a store built out of that same file proves + nothing, and `bin/perry-tasks` learned that the hard way: a planted + hand edit passed, because both sides saw the edited value. + """ + p = Project(self) + for cmd in ("render", "diff", "verify"): + with self.subTest(cmd): + proc = p.okr(cmd) + self.assertEqual(proc.returncode, 2, proc.stdout) + self.assertIn("no store on disk yet", proc.stderr) + + def test_write_requires_the_explicit_import_flag(self): + p = Project(self) + proc = p.okr("write") + self.assertEqual(proc.returncode, 1) + self.assertIn("--from-file", proc.stderr) + + def test_the_full_cycle_is_byte_identical_on_both_files(self): + p = Project(self) + for tool, text_of in ((p.okr, p.okr_text), (p.config, p.config_text)): + before = text_of() + self.assertEqual(tool("write", "--from-file").returncode, 0) + diff = tool("diff") + self.assertEqual(diff.returncode, 0, diff.stdout) + report = json.loads(diff.stdout) + self.assertTrue(report["identical"]) + self.assertEqual(report["cells_verbatim"], {}) + self.assertEqual(tool("verify").returncode, 0) + # `render` without `--write` prints and touches nothing. + rendered = tool("render") + self.assertEqual(rendered.returncode, 0, rendered.stderr) + self.assertEqual(rendered.stdout, before) + self.assertEqual(text_of(), before) + + def test_render_write_puts_a_drifted_file_back_in_line(self): + p = Project(self) + p.okr("write", "--from-file") + before = p.okr_text() + (p.root / "perry" / "OKR.md").write_text( + before.replace("| KR-O1.1 |", "| KR-O1.1 |", 1) + .replace("3 of 3 modes live", "SEVEN of 3 modes live")) + self.assertEqual(p.okr("diff").returncode, 1) + self.assertEqual(p.okr("render", "--write").returncode, 0) + self.assertEqual(p.okr_text(), before) + + +class TestAHandEditIsReportedAndNeitherHonouredNorOverwritten( + unittest.TestCase): + """V4 step 5 — the contract `perry-tasks diff` gives the board. + + Three separate claims, and the middle one is the one a renderer usually + gets wrong by being helpful: + + REPORTED `diff` exits non-zero and NAMES the cell. + not honoured `write` refuses rather than replacing the store's + canonical value with what the file happens to say. + not overwritten reading the file — `diff`, `verify`, `render` without + `--write` — leaves every byte of the edit in place. + """ + + def setUp(self): + self.p = Project(self) + self.p.okr("write", "--from-file") + self.p.config("write", "--from-file") + + def test_an_okr_hand_edit(self): + path = self.p.root / "perry" / "OKR.md" + path.write_text(path.read_text().replace( + "| 3 of 3 modes live |", "| two of three, honestly |", 1)) + + diff = self.p.okr("diff") + self.assertEqual(diff.returncode, 1) + report = json.loads(diff.stdout) + self.assertFalse(report["identical"]) + drift = report["cells_the_store_and_the_file_disagree_on"] + self.assertEqual(len(drift), 1, drift) + self.assertEqual(drift[0]["column"], "Metric / Target") + self.assertEqual(drift[0]["file"], "two of three, honestly") + self.assertIn("KR-O1.1", drift[0]["key"]) + + write = self.p.okr("write", "--from-file") + self.assertEqual(write.returncode, 1) + self.assertIn("refusing to overwrite", write.stderr) + self.assertIn("two of three, honestly", write.stderr) + + self.assertIn("two of three, honestly |", path.read_text()) + self.assertIn("3 of 3 modes live", + (self.p.root / "perry" / "okr.jsonl").read_text()) + + def test_an_appended_hand_edit_is_counted_rather_than_hidden(self): + """The edit `diff` alone calls identical, and the reason it does. + + `describe_cell` keeps whatever sits around the stored value as + presentation — that is what makes `~~**ALLOC-01**~~` a struck-through + id rather than a different id. An edit that APPENDS to a cell rides + the same branch: the stored value is still in there, the extra words + are kept as a suffix, and the file renders back byte for byte. So the + bytes cannot report it and something else has to. `verify` is where it + surfaces, and `cells_wearing_decoration` is the count — measured 0 + across every file in this repository, which is what makes a non-zero + one worth reading. + """ + path = self.p.root / "perry" / "OKR.md" + path.write_text(path.read_text().replace( + "| 3 of 3 modes live |", "| 3 of 3 modes live, honest |", 1)) + + diff = json.loads(self.p.okr("diff").stdout) + self.assertTrue(diff["identical"]) + self.assertEqual(diff["cells_wearing_decoration"], + {"Metric / Target": 1}) + + verify = self.p.okr("verify") + self.assertEqual(verify.returncode, 1, verify.stdout) + self.assertEqual(json.loads(verify.stdout)["cells_wearing_decoration"], + {"Metric / Target": 1}) + + write = self.p.okr("write", "--from-file") + self.assertEqual(write.returncode, 1) + self.assertIn("3 of 3 modes live, honest", write.stderr) + + def test_a_config_hand_edit(self): + path = self.p.root / ".perry" / "config.md" + path.write_text(path.read_text().replace( + "- Repo layout: single", "- Repo layout: split")) + + diff = self.p.config("diff") + self.assertEqual(diff.returncode, 1) + drift = json.loads(diff.stdout)["cells_the_store_and_the_file_disagree_on"] + self.assertEqual([(d["key"], d["file"], d["store"]) for d in drift], + [("setting/repo_layout", "split", "single")]) + + write = self.p.config("write", "--from-file") + self.assertEqual(write.returncode, 1) + self.assertIn("setting/repo_layout", write.stderr) + self.assertIn("- Repo layout: split", path.read_text()) + + def test_a_deleted_line_is_reported_rather_than_dropped(self): + """The edit `cmp` alone would call a smaller file. + + A row that leaves the file is a record with nowhere to render. That is + a hole in the projection and it has to be named, because nothing in a + byte comparison distinguishes it from a shorter document. + """ + path = self.p.root / ".perry" / "config.md" + path.write_text("\n".join( + l for l in path.read_text().split("\n") + if not l.startswith("- Chat language:"))) + report = json.loads(self.p.config("diff").stdout) + self.assertIn("setting/chat_language", + report["records_not_in_the_file"]) + + +class TestTheReadContractsDoNotMove(unittest.TestCase): + """V4's fifth deliverable: a consumer pinned to today's payload needs no edit. + + This row changes where the bytes come from, not what any reader is told. + """ + + def test_perry_goals_list_is_identical_before_and_after_the_store_exists(self): + p = Project(self) + before = run("perry-goals", "list", "--json", root=p.root) + self.assertEqual(before.returncode, 0, before.stderr) + self.assertEqual(p.okr("write", "--from-file").returncode, 0) + after = run("perry-goals", "list", "--json", root=p.root) + self.assertEqual(after.returncode, 0, after.stderr) + + a, b = json.loads(before.stdout), json.loads(after.stdout) + for payload in (a, b): + payload.pop("project_root", None) + payload.pop("state_root", None) + self.assertEqual(a, b, "minting the store moved the read contract") + self.assertGreater(len(a["krs"]), 0) + + def test_the_store_holds_every_kr_the_shipped_reader_reads(self): + """Two readers of one file, compared rather than trusted. + + `viewer/parsers.py` reads the CURRENT version block only; the store + holds every version. So the assertion is containment in that + direction — and a KR the shipped reader sees that the store does not + hold would be a row silently dropped on the way into the store. + """ + text = (ROOT / "perry" / "OKR.md").read_text() + okr = P.parse_okr(text) + read = {k.id for o in okr.objectives for k in o.krs} + # `okr.version` is the `## v: ` heading text, which is exactly + # what the store files each KR under — so the two are asking about the + # same block and set equality is the right assertion. (The objective + # title is NOT comparable: the parser cleans `Objective 1 — …` down to + # the title, and the store keys on the heading as authored.) + held = {r["id"] for r in M.derive(M.OKR, text) + if r["kind"] == "kr" and r["version"] == okr.version} + self.assertTrue(read, "the fixture parsed to no KRs at all") + self.assertEqual( + read, held, + "the shipped reader and the store disagree about which KRs the " + "current version block holds") + + +class TestTheColumnSetsComeFromTheSchema(unittest.TestCase): + """One declaration of which columns exist, not two. + + `perry-lint` validates `## Commitments` and `## Tracks` against + `schema/state-schema.json`. A second list here would disagree with it the + day a column is added — and disagree in silence, because an unknown column + renders verbatim and the file still passes `cmp`. + """ + + def test_the_maps_are_read_rather_than_restated(self): + self.assertEqual( + M.COMMITMENT_COLUMNS, + M.table_columns("OKR.md", "Commitments")) + self.assertEqual( + M.TRACK_COLUMNS, + M.table_columns(".perry/config.md", "Tracks")) + # And they resolve to the keys `bin/perry-state` files a track under. + self.assertEqual(set(M.TRACK_COLUMNS.values()), + {"track", "mode", "spine", "stages", "wip", "sla", + "cycle", "default_rung"}) + + def test_a_declared_column_with_no_store_field_is_refused_at_import(self): + """Guard against the guard. + + `record` copies `STORED[kind]` and nothing else, so a column read into + a site and absent from `STORED` would be dropped in silence. The check + is asserted here by taking the field away and watching it fire, which + is the only way to know it can. + """ + original = M.STORED["track"] + M.STORED["track"] = tuple(f for f in original if f != "sla") + try: + with self.assertRaises(M.Refused) as caught: + M._assert_every_declared_column_is_stored() + self.assertIn("'sla'", str(caught.exception)) + finally: + M.STORED["track"] = original + # And it passes as shipped. + M._assert_every_declared_column_is_stored() + + def test_the_tracks_heading_is_the_schemas_own(self): + """`^Tracks\\b|^轨道`, read from the file that declares it. + + Written out here it would be the second copy — `bin/perry-state § + parse_tracks` holds the first — and the Chinese half is exactly the + kind of alternative a hand-copy loses. + """ + pattern = M.config_table_under("Tracks") + self.assertTrue(pattern.match("Tracks")) + self.assertTrue(pattern.match("轨道")) + self.assertFalse(pattern.match("Notes")) + + +class TestTheWriterWritesTheStore(unittest.TestCase): + """Deliverable 3 — `perry-goals`' write path targets the store. + + The register is `## Commitments`, which Perry's own `OKR.md` does not + carry, so the fixture is the second project's: a `pipeline` track, a + `queue` track, and a register with two live rows. + """ + + def project(self) -> pathlib.Path: + d = pathlib.Path(tempfile.mkdtemp(prefix="perry-goals-store-")) + self.addCleanup(shutil.rmtree, d, ignore_errors=True) + shutil.copytree(FIXTURES / "second-project", d, dirs_exist_ok=True) + cfg = d / ".perry" / "config.md" + cfg.write_text(cfg.read_text() + GATE_OFF, encoding="utf-8") + return d + + def test_commit_writes_okr_and_the_store_together(self): + d = self.project() + store = d / "okr.jsonl" + self.assertFalse(store.exists()) + proc = run("perry-goals", "commit", "--track", "ops", + "--promise", "Reconcile the July statement", "--to", "RM", + "--due", "2026-09-30", "--json", root=d) + self.assertEqual(proc.returncode, 0, proc.stderr) + result = json.loads(proc.stdout) + cid = result["id"] + + self.assertTrue(store.exists(), "the write did not mint the store") + records = M.load_store(store) + row = next(r for r in records + if r["kind"] == "commitment" and r["id"] == cid) + self.assertEqual(row["promise"], "Reconcile the July statement") + self.assertEqual(row["to_whom"], "RM") + self.assertEqual(row["due"], "2026-09-30") + self.assertEqual(row["status"], "active") + + # And the file is now a projection of it, byte for byte. + diff = run("perry-okr", "diff", root=d) + self.assertEqual(diff.returncode, 0, diff.stdout) + self.assertTrue(json.loads(diff.stdout)["identical"]) + + def test_a_second_commit_keeps_the_projection_exact(self): + d = self.project() + for n in range(2): + proc = run("perry-goals", "commit", "--track", "research", + "--promise", f"Memo {n}", "--to", "用户", + "--due", f"2026-1{n}-01", "--json", root=d) + self.assertEqual(proc.returncode, 0, proc.stderr) + diff = run("perry-okr", "diff", root=d) + self.assertEqual(diff.returncode, 0, diff.stdout) + report = json.loads(diff.stdout) + self.assertTrue(report["identical"]) + self.assertEqual(report["cells_verbatim"], {}) + self.assertEqual(report["kinds"]["commitment"], 4) + + def test_a_hand_edit_is_reported_by_the_writer_and_not_swallowed(self): + """Perry's own Operating Principle: *a hand edit is reported, never + refused*. So the writer says so on stderr and proceeds — it does not + invent a second refusal beside `check_hand_edit`, and it does not + absorb the edit in silence, which is the defect `bin/perry-tasks § + write` records.""" + d = self.project() + run("perry-goals", "commit", "--track", "ops", "--promise", "a", + "--to", "RM", "--due", "3d", root=d) + okr = d / "OKR.md" + okr.write_text(okr.read_text().replace("Weekly candidate memo", + "Weekly candidate memo (revised)")) + proc = run("perry-goals", "commit", "--track", "ops", "--promise", "b", + "--to", "RM", "--due", "3d", root=d) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("edited by hand", proc.stderr) + self.assertIn("research/1", proc.stderr) + # Proceeded, and the projection is exact again. + self.assertEqual(run("perry-okr", "diff", root=d).returncode, 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/fixtures/live-state/track_attribution.before.py b/tests/fixtures/live-state/track_attribution.before.py new file mode 100644 index 00000000..dd3dc1cc --- /dev/null +++ b/tests/fixtures/live-state/track_attribution.before.py @@ -0,0 +1,176 @@ +"""Evidence belongs to the track that produced it, not to whichever is first. + +`bin/perry-diagnose` used `len(tracks) == 1` as a proxy for **"no register +exists"**. It is also true when the register **declares exactly one track** — +and then every board row, every commitment row and every project-wide file was +scored as that track's, ignoring the `Track` column the scanner already reads +and ignoring `schema/state-schema.json`'s own rule that a blank `Track` means +the implicit `main` track. + +A project declaring one `pipeline` track with its ordinary work in untracked +rows scored `project 7 / pipeline 4` and got a `MODE-01` warn **telling the +user to change a `Mode` cell that was correct**, citing objectives and rows +that are not that track's. `perry-lint` accepted the shape and no test covered +it. + +Found by a reviewer that had already fixed attribution across *modes* and then +asked whether attribution across *tracks* had ever been examined. It had not. + +Run: python3 tests/parallel test_track_attribution +""" + +from __future__ import annotations + +import json +import pathlib +import shutil +import subprocess +import sys +import tempfile +import unittest + +ROOT = pathlib.Path(__file__).resolve().parent.parent +TOOL = ROOT / "bin" / "perry-diagnose" + +BOARD = """# Board + +## P1 + +| ID | Title | Owner | Status | Next action | Evidence | Verification | Track | Stage | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| TASK-001 | ordinary work | C | not_started | — | — | V2 | | | +| TASK-002 | more of it | C | not_started | — | — | V2 | | | +| TASK-003 | a pipeline item | C | not_started | — | — | V2 | ops | draft | +""" + +TRACKED_BOARD = """# Board + +## P1 + +| ID | Title | Owner | Status | Next action | Evidence | Verification | Track | Stage | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| TASK-003 | a pipeline item | C | not_started | — | — | V2 | ops | draft | +""" + +UNTRACKED_COMMITMENT = """# OKR + +## Objectives + +- O1 ship it + +## Commitments + +| Id | Track | Promise | Due | By when note | +|---|---|---|---|---| +| COM-001 | | standing work | — | monthly | +""" + +REGISTER = """# Config + +State root: perry + +## Tracks + +| Track | Mode | Spine | Stages | WIP | SLA | Cycle | Default rung | +|---|---|---|---|---|---|---|---| +| ops | pipeline | OKR.md | brief,draft | — | 3d | — | V2 | +""" + + +class TrackCase(unittest.TestCase): + def project(self, register: str | None, board: str = BOARD, + okr: str = "# OKR\n\n## Objectives\n\n- O1 ship it\n"): + d = pathlib.Path(tempfile.mkdtemp()) + self.addCleanup(shutil.rmtree, d, ignore_errors=True) + (d / "perry").mkdir() + (d / ".perry").mkdir() + (d / ".perry" / "config.md").write_text( + register if register else "# Config\n\nState root: perry\n") + (d / "perry" / "BOARD.md").write_text(board) + (d / "perry" / "OKR.md").write_text(okr) + (d / "perry" / "phase").mkdir() + (d / "perry" / "phase" / "001-a.md").write_text("# Phase 1\n") + return d + + def modes(self, register, board: str = BOARD, + okr: str = "# OKR\n\n## Objectives\n\n- O1 ship it\n"): + proc = subprocess.run( + [sys.executable, str(TOOL), "--root", + str(self.project(register, board, okr)), + "--json"], capture_output=True, text=True, cwd=ROOT) + self.assertEqual(proc.returncode, 0, proc.stderr[-400:]) + w = json.loads(proc.stdout)["work_modes"] + return {t["track"]: t for t in w["tracks"]}, w + + +class TestADeclaredTrackIsNotGivenEverything(TrackCase): + def test_the_declared_track_is_scored_on_its_own_rows(self): + tracks, _ = self.modes(REGISTER) + ops = tracks["ops"] + self.assertEqual(ops["mode"], "pipeline") + self.assertEqual(ops["scores"]["project"], 0, + "the project's own objectives and phases were " + "counted as this track's evidence") + + def test_the_implicit_main_track_is_enumerated(self): + """Rows with a blank `Track` belong to `main` by the schema's rule, and + `main` was in nobody's list — so their evidence went to the declared + track or nowhere.""" + tracks, _ = self.modes(REGISTER) + self.assertIn("main", tracks) + self.assertFalse(tracks["main"]["declared"]) + self.assertEqual(tracks["main"]["mode"], "project") + + def test_project_wide_files_go_to_the_project_wide_track(self): + """`phase/` and `OKR.md`'s objectives describe the whole repository. + Fixing the first bug sent them to NOBODY, which is the same error + pointed the other way.""" + tracks, _ = self.modes(REGISTER) + self.assertGreater(tracks["main"]["scores"]["project"], 0) + + +class TestNoRegisterBehavesExactlyAsBefore(TrackCase): + def test_one_implicit_track_gets_everything(self): + """The unchanged path, and the one every existing project is on. A fix + that moved this would be a regression dressed as a correction.""" + tracks, w = self.modes(None) + self.assertFalse(w["register_declared"]) + self.assertEqual(list(tracks), ["main"]) + self.assertGreater(tracks["main"]["scores"]["project"], 0) + + +class TestCommitmentsAlsoBelongToATrack(TrackCase): + def test_an_untracked_commitment_enumerates_implicit_main(self): + tracks, _ = self.modes(REGISTER, TRACKED_BOARD, + UNTRACKED_COMMITMENT) + self.assertIn("main", tracks) + self.assertFalse(tracks["main"]["declared"]) + self.assertTrue(any( + "standing commitment" in item + for item in tracks["main"]["evidence"]["queue"])) + + def test_repository_evidence_does_not_accuse_the_declared_pipeline(self): + tracks, _ = self.modes(REGISTER, TRACKED_BOARD, + UNTRACKED_COMMITMENT) + self.assertEqual(tracks["ops"]["scores"]["project"], 0) + self.assertEqual(tracks["ops"]["mode"], "pipeline") + + def test_a_sole_non_project_track_does_not_inherit_repository_evidence(self): + tracks, _ = self.modes(REGISTER, TRACKED_BOARD) + self.assertNotIn("main", tracks) + self.assertEqual(tracks["ops"]["scores"]["project"], 0) + self.assertEqual(tracks["ops"]["mode"], "pipeline") + + +class TestPerrysOwnProjectIsUnmoved(unittest.TestCase): + def test_it_still_reads_one_project_track(self): + proc = subprocess.run( + [sys.executable, str(TOOL), "--json"], + capture_output=True, text=True, cwd=ROOT) + w = json.loads(proc.stdout)["work_modes"] + self.assertFalse(w["register_declared"]) + self.assertEqual([t["track"] for t in w["tracks"]], ["main"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/fixtures/live-state/v5_signoff.before.py b/tests/fixtures/live-state/v5_signoff.before.py new file mode 100644 index 00000000..1077e304 --- /dev/null +++ b/tests/fixtures/live-state/v5_signoff.before.py @@ -0,0 +1,581 @@ +"""TASK-109 — a V5 sign-off is SELECTED from measured facts, not authored. + +V5 is the one rung whose content is a human's: "name, date, and what they +checked". Until this row the tool took none of it. `done --rung V5` wrote a +rung, and the signature was a paragraph the user composed by hand. + +TASK-047 is the case that named the defect. Perry ran three checks, printed +their output, and showed the user; the user then wrote, from memory, a sentence +describing those same three checks. Two things are wrong with that, and the +second is the one that matters: + + 1. the user re-derives by hand a record the system already holds, and + 2. free text cannot distinguish *I re-ran this* from *Perry ran this and I + read the output*. The gap widens the more Perry does. + +What keeps the fix from being a rubber stamp is one rule, and it is enforced +here mechanically rather than by review: + + Perry may draft only facts it MEASURED. It may never draft a claim about + what the USER did. + +`TestNoDraftedOptionAssertsAUserAction` is that rule as a test over the option +builder — the only place a drafted string is minted. + +Run: python3 -m unittest discover -s tests (or ./tests/run) +""" + +from __future__ import annotations + +import importlib.machinery +import importlib.util +import json +import re +import subprocess +import tempfile +import unittest +from datetime import date +from pathlib import Path + +from gate import GATE_OFF # tests/gate.py — why this fixture opts out + +PERRY_HOME = Path(__file__).resolve().parent.parent +TOOL = PERRY_HOME / "bin" / "perry-task" +TASKS = PERRY_HOME / "bin" / "perry-tasks" + + +def load_tool(): + spec = importlib.util.spec_from_loader( + "perry_task", importlib.machinery.SourceFileLoader("perry_task", str(TOOL))) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +PT = load_tool() + +BOARD = """# Board — T + +## P0 (must finish this period) + +| ID | Title | Owner | Status | Next action | Evidence | +|---|---|---|---|---|---| + +## P1 + +| ID | Title | Owner | Status | Next action | Evidence | +|---|---|---|---|---|---| + +## P2 + +| ID | Title | Owner | Status | Next action | Evidence | +|---|---|---|---|---|---| +""" + +#: The fixture V5 verification 1 asks for: three facts Perry measured during +#: the task, and one claim it is only passing along. Worded the way a real +#: dispatch would word them — an observation, never an act. +MEASURED = [ + "claims[] carries zero changed lines in the diff on schema/state-schema.json", + "write behaviour under enforce / advisory / declared: 3 runs, exit codes recorded", + "the perry-migrate exemption still runs on an undeclared project", +] +RESTATED = ["the branch carrying this change is unmerged"] + + +class Project: + """A throwaway Perry project the tool can close a row in.""" + + def __init__(self): + self.dir = tempfile.TemporaryDirectory() + self.root = Path(self.dir.name) + (self.root / ".perry").mkdir() + (self.root / ".perry" / "config.md").write_text( + "# Perry configuration\n\n- Document language: English\n" + "- Repo layout: single\n- State root: .\n" + GATE_OFF) + (self.root / "BOARD.md").write_text(BOARD) + r = subprocess.run( + ["python3", str(TASKS), "write", "--from-board", "--root", + str(self.root)], capture_output=True, text=True) + if r.returncode: + raise AssertionError(r.stdout + r.stderr) + + def run(self, *argv) -> tuple[int, dict | str]: + r = subprocess.run( + ["python3", str(TOOL), *argv, "--root", str(self.root), "--json"], + capture_output=True, text=True) + try: + return r.returncode, json.loads(r.stdout or "{}") + except json.JSONDecodeError: + return r.returncode, r.stdout + r.stderr + + def a_task(self) -> str: + _, a = self.run("add", "--title", "Flip the conformance default", + "--deliverable", "the gate enforces", + "--verification", "the suite is green") + return a["id"] + + def journal(self) -> str: + for p in (self.root / "journal").rglob("*.md"): + return p.read_text() + return "" + + def events(self) -> list[dict]: + p = self.root / ".perry" / "events.jsonl" + return ([json.loads(l) for l in p.read_text().split("\n") if l.strip()] + if p.exists() else []) + + def close_v5(self, tid, *extra): + return self.run("done", tid, "--evidence", "evidence/x.md", + "--rung", "V5", + *[f for t in MEASURED for f in ("--measured", t)], + *[f for t in RESTATED for f in ("--restated", t)], + *extra) + + def __del__(self): + self.dir.cleanup() + + +class TestTheRecordIsASelection(unittest.TestCase): + """V5 verification 1 — three measured items, one restated, two selected.""" + + def setUp(self): + self.p = Project() + self.tid = self.p.a_task() + self.code, self.out = self.p.close_v5(self.tid, "--checked", "1,3") + self.assertEqual(self.code, 0, self.out) + self.sig = self.out["signoff"] + + def test_the_two_selected_are_recorded_as_checked(self): + checked = [i["text"] for i in self.sig["items"] + if i["disposition"] == "checked"] + self.assertEqual(checked, [MEASURED[0], MEASURED[2]]) + + def test_the_two_unselected_are_recorded_as_accepted_on_report(self): + """Deliverable 4. Not dropped — that is the whole gain over free text, + which could not distinguish the two at all.""" + rest = [i["text"] for i in self.sig["items"] + if i["disposition"] == "accepted on report"] + self.assertEqual(rest, [MEASURED[1], RESTATED[0]]) + self.assertEqual(len(self.sig["items"]), 4, + "an offered item left the record entirely") + + def test_the_labels_survive_verbatim_into_the_written_record(self): + """Both labels, in the journal, spelled exactly as the record spells + them. A record whose markdown says "accepted" and whose JSON says + `accepted on report` has two answers to the question the rung asks.""" + journal = self.p.journal() + self.assertIn("## V5 sign-off", journal) + for label in ("checked", "accepted on report"): + self.assertIn(f"**{label}**", journal, + f"the disposition label {label!r} did not survive") + for item in self.sig["items"]: + self.assertIn(item["text"], journal) + self.assertIn("*(Perry verified)*", journal) + self.assertIn("*(restated — Perry did not verify this)*", journal) + + def test_every_item_is_labelled_with_its_provenance(self): + """Deliverable 2. Selecting a Perry-verified item means *I checked this + too*; selecting a restated one means *I checked a claim Perry passed + along*. The label is what keeps those from reading the same.""" + by_text = {i["text"]: i["provenance"] for i in self.sig["items"]} + for text in MEASURED: + self.assertEqual(by_text[text], "Perry verified") + self.assertEqual(by_text[RESTATED[0]], + "restated — Perry did not verify this") + + def test_name_and_date_are_filled_in_not_typed(self): + """Deliverable 3 — the two fields a human should never be retyping.""" + self.assertTrue(self.sig["signed_by"].strip(), + "an anonymous signature is not a weaker signature") + self.assertEqual(self.sig["signed_on"], f"{date.today():%Y-%m-%d}") + self.assertIn(self.sig["signed_by"], self.p.journal()) + + def test_the_signature_rides_in_the_event_too(self): + done = [e for e in self.p.events() if e.get("event") == "done"] + self.assertEqual(len(done), 1) + self.assertEqual(done[0]["signoff"]["counts"], + {"checked": 2, "accepted on report": 2, + "not looked at": 0}) + + def test_the_row_still_closes_at_v5(self): + """The sign-off is added to the close; it does not replace it.""" + self.assertEqual(self.out["rung"], "V5") + self.assertNotIn(self.tid, (self.p.root / "BOARD.md").read_text()) + + +class TestFreeTextIsAdditive(unittest.TestCase): + """V5 verification 2 — recorded ALONGSIDE the selection, never instead.""" + + ALSO = "I re-ran perry-conform declare --all against my own checkout." + + def test_free_text_lands_beside_the_selection(self): + p = Project() + tid = p.a_task() + code, out = p.close_v5(tid, "--checked", "2", "--also", self.ALSO) + self.assertEqual(code, 0, out) + sig = out["signoff"] + self.assertEqual(sig["also_checked"], self.ALSO) + self.assertEqual(len(sig["items"]), 4, + "free text replaced the selection instead of joining it") + self.assertEqual(sig["counts"]["checked"], 1) + journal = p.journal() + self.assertIn(self.ALSO, journal) + self.assertIn(MEASURED[1], journal) + self.assertIn("**accepted on report**", journal) + + def test_free_text_alone_is_a_signature(self): + """Deliverable 5 read the other way: the user may have checked only + something Perry never offered. That is a signature, not an empty one.""" + p = Project() + tid = p.a_task() + code, out = p.close_v5(tid, "--checked", "none", "--also", self.ALSO) + self.assertEqual(code, 0, out) + self.assertEqual(out["signoff"]["counts"]["checked"], 0) + self.assertEqual(out["signoff"]["also_checked"], self.ALSO) + + def test_the_free_text_is_not_run_through_the_drafting_guard(self): + """`--also` is the USER's sentence. The guard exists to stop PERRY + drafting a claim about a person; applying it to the user's own words + would refuse them for describing what they did, which is the one thing + only they may say.""" + p = Project() + tid = p.a_task() + code, out = p.close_v5( + tid, "--checked", "none", + "--also", "I reviewed the diff myself and approved it.") + self.assertEqual(code, 0, out) + self.assertIn("I reviewed the diff myself", p.journal()) + + +class TestNoDraftedOptionAssertsAUserAction(unittest.TestCase): + """V5 verification 3 and deliverable 6, as a machine check. + + This is the load-bearing test in the file. If Perry may draft *the user + reviewed the diff*, then a V5 close is Perry certifying its own work with a + human's name on it, and every other guarantee here is decoration. + + A review comment cannot enforce this: it holds until the first hurried + close. `PT.signoff_options` is the only place a drafted string is minted, + so the rule is checked there and nowhere else needs to remember it. + """ + + #: Each of these is a claim about a person. None is Perry's to write. + USER_CLAIMS = [ + "the user reviewed the diff", + "the user accepted the two costs", + "you confirmed the migrate exemption", + "your checkout was declared", + "the human read the fixture opt-out reasoning", + "the reviewer signed off on the rung", + "the signer approved the enforce default", + "reviewed the claims[] diff", + "approved the migration plan", + "accepted on the strength of the printed output", + "acknowledged the residue on a real board", + "用户已阅并接受两项代价", + "同意把 enforce 设为默认值", + ] + + #: The complement, and the anti-vacuity guard: a rule that refused + #: everything would pass the list above and be useless. Every one of these + #: is a fact Perry can measure, and several are near-misses on purpose — + #: `user-facing` contains `user`, `acceptance` contains `accept`. + MEASURABLE = [ + "claims[] carries zero changed lines in the diff", + "the perry-migrate exemption still runs on an undeclared project", + "the user-facing message names the mode rather than the literal string", + "3 of 80 closed rows carry V5", + "the acceptance-criteria file resolves to an existing path", + "tests/parallel: 59 modules, 1717 tests, 3 red", + "SKILL.md is 21030 bytes against a 20480 cap", + "owner is present on 21 of 21 open rows and 0 of 60 closed ones", + ] + + def test_a_drafted_claim_about_a_person_is_refused(self): + for claim in self.USER_CLAIMS: + for flag, kwargs in (("--measured", {"measured": [claim]}), + ("--restated", {"restated": [claim]})): + with self.subTest(claim=claim, flag=flag): + with self.assertRaises(PT.Refused) as caught: + PT.signoff_options(kwargs.get("measured", []), + kwargs.get("restated", [])) + self.assertIn(claim, str(caught.exception), + "the refusal must quote what it refused") + + def test_a_measured_fact_is_not_refused(self): + options = PT.signoff_options(self.MEASURABLE, []) + self.assertEqual(len(options), len(self.MEASURABLE)) + self.assertEqual([o["n"] for o in options], + list(range(1, len(self.MEASURABLE) + 1))) + + def test_the_refusal_reaches_the_cli_not_just_the_function(self): + p = Project() + tid = p.a_task() + code, out = p.run("done", tid, "--evidence", "e.md", "--rung", "V5", + "--measured", "the user reviewed the diff", + "--checked", "1") + self.assertEqual(code, 1) + self.assertIn("refused", out) + self.assertEqual(p.journal().count("V5 sign-off"), 0, + "a refused sign-off wrote something anyway") + + def test_no_option_the_builder_emits_carries_a_user_claim(self): + """The rule stated over the OUTPUT rather than the input, so a future + builder that rewrites or decorates an option cannot smuggle one past + the entry check.""" + for option in PT.signoff_options(self.MEASURABLE, ["a restated claim"]): + PT.check_no_user_claim(option["text"], "--measured") + self.assertIn(option["provenance"], + ("Perry verified", + "restated — Perry did not verify this")) + + +class TestAnEmptySignatureIsRefused(unittest.TestCase): + """V5 verification 4 and deliverable 7.""" + + def test_nothing_selected_and_no_free_text_is_refused(self): + p = Project() + tid = p.a_task() + code, out = p.close_v5(tid, "--checked", "none") + self.assertEqual(code, 1, out) + self.assertIn("not a sign-off", json.dumps(out, ensure_ascii=False)) + + def test_the_refused_close_wrote_nothing(self): + """A refusal that half-closed the row would be worse than the blank + signature it prevented.""" + p = Project() + tid = p.a_task() + p.close_v5(tid, "--checked", "none") + self.assertIn(tid, (p.root / "BOARD.md").read_text()) + self.assertEqual([e["event"] for e in p.events()], ["add"]) + + def test_pressing_return_at_the_prompt_is_what_this_costs(self): + """`--checked` absent entirely is the same keystroke as `none`, and + must not be the cheap path to a blank signature.""" + p = Project() + tid = p.a_task() + code, _ = p.close_v5(tid) + self.assertEqual(code, 1) + + def test_an_empty_offered_item_is_refused(self): + with self.assertRaises(PT.Refused): + PT.signoff_options([""], []) + + def test_a_bare_close_that_never_engaged_the_path_is_unchanged(self): + """The refusal fires when the sign-off path was ENGAGED and produced + nothing. A close that passes no sign-off flag at all writes no + signature rather than a blank one — rungs are advisory this release + (DESIGN-003 § 4 decision 4) and hardening the rung itself is out of + this row's scope.""" + p = Project() + tid = p.a_task() + code, out = p.run("done", tid, "--evidence", "e.md", "--rung", "V5") + self.assertEqual(code, 0, out) + self.assertIsNone(out["signoff"]) + + +class TestThreeDispositionsNotTwo(unittest.TestCase): + """The subjective question this row was dispatched with, pinned. + + **The alternative that was rejected: two categories** — `checked` and + `accepted on report` — with nothing between "I read Perry's output and took + its word" and "I never looked at this at all". + + It was rejected on the corpus. All three V5 signatures already in this + repository write the third category by hand. TASK-034's carries a section + headed *"Not checked, and recorded because V5's whole value is saying so"* + beside what it did check. TASK-047's distinguishes *fixture opt-out 的理由已读 + 并接受* — read, then accepted — from two costs taken on the strength of + Perry's printed output. A format that cannot hold what the existing + signatures already say is a regression against the corpus it must stay + compatible with. + + The second reason is the drafting rule. Defaulting an unselected item to + `accepted on report` is already the outer edge of what Perry may assert: + it describes the SCOPE of the signature, not an act the user performed. So + `not looked at` is never a default — it is reachable only by the user + naming the item, which is what keeps it a user statement. + """ + + def test_not_looked_at_is_a_disposition_of_its_own(self): + p = Project() + tid = p.a_task() + code, out = p.close_v5(tid, "--checked", "1", "--not-looked-at", "4") + self.assertEqual(code, 0, out) + by_text = {i["text"]: i["disposition"] for i in out["signoff"]["items"]} + self.assertEqual(by_text[MEASURED[0]], "checked") + self.assertEqual(by_text[MEASURED[1]], "accepted on report") + self.assertEqual(by_text[RESTATED[0]], "not looked at") + self.assertIn("**not looked at**", p.journal()) + + def test_not_looked_at_is_never_a_default(self): + p = Project() + tid = p.a_task() + _, out = p.close_v5(tid, "--checked", "1") + self.assertEqual(out["signoff"]["counts"]["not looked at"], 0) + + def test_one_item_cannot_carry_two_dispositions(self): + p = Project() + tid = p.a_task() + code, _ = p.close_v5(tid, "--checked", "1", "--not-looked-at", "1") + self.assertEqual(code, 1) + + +class TestTheOfferIsBuiltNotComposed(unittest.TestCase): + """Deliverable 1, and the numbering contract between offer and close.""" + + def test_the_offer_numbers_items_the_way_done_reads_them(self): + p = Project() + tid = p.a_task() + code, offer = p.run( + "signoff-offer", tid, + *[f for t in MEASURED for f in ("--measured", t)], + *[f for t in RESTATED for f in ("--restated", t)]) + self.assertEqual(code, 0, offer) + self.assertEqual([o["text"] for o in offer["options"]], + MEASURED + RESTATED) + _, out = p.close_v5(tid, "--checked", "3") + picked = [i["text"] for i in out["signoff"]["items"] + if i["disposition"] == "checked"] + self.assertEqual(picked, [offer["options"][2]["text"]], + "option 3 in the prompt is not option 3 in the record") + + def test_the_offer_writes_nothing(self): + p = Project() + tid = p.a_task() + before = (p.root / "BOARD.md").read_text() + p.run("signoff-offer", tid, "--measured", MEASURED[0]) + self.assertEqual((p.root / "BOARD.md").read_text(), before) + self.assertEqual([e["event"] for e in p.events()], ["add"]) + + def test_it_degrades_to_a_numbered_free_text_prompt(self): + """`reference/host-capabilities.md § Prompt rendering`: Codex has no + selection UI and gets numbered options plus free text. The RENDERING + changes per host; the record does not.""" + p = Project() + tid = p.a_task() + _, offer = p.run("signoff-offer", tid, + *[f for t in MEASURED for f in ("--measured", t)]) + self.assertTrue(offer["multi_select"]) + prompt = offer["prompt"] + for n in (1, 2, 3): + self.assertIn(f" {n}) ", prompt) + self.assertIn("all", prompt) + self.assertIn("none", prompt) + self.assertIn("accepted on report", prompt) + self.assertIn(tid, prompt) + + def test_the_same_selection_records_the_same_thing_from_either_spelling(self): + """`--checked 1,3` is what the free-text host hands back; `--checked 1 + --checked 3` is what a structured host produces. One record.""" + a, b = Project(), Project() + _, one = a.close_v5(a.a_task(), "--checked", "1,3") + _, two = b.close_v5(b.a_task(), "--checked", "1", "--checked", "3") + strip = lambda s: [(i["n"], i["disposition"]) for i in s["signoff"]["items"]] + self.assertEqual(strip(one), strip(two)) + + def test_an_offer_with_nothing_measured_is_refused(self): + p = Project() + tid = p.a_task() + code, _ = p.run("signoff-offer", tid) + self.assertEqual(code, 1) + + def test_a_signoff_on_a_rung_that_is_not_v5_is_refused(self): + """V5 is "human sign-off"; V4 is a rubric and V6 is the world. Hanging + a signature on either records a human gate nobody asked for.""" + for rung in ("V3", "V4", "V6"): + with self.subTest(rung=rung): + p = Project() + code, _ = p.run("done", p.a_task(), "--evidence", "e.md", + "--rung", rung, "--measured", MEASURED[0], + "--checked", "1") + self.assertEqual(code, 1) + + +class TestHistoryIsNotRewritten(unittest.TestCase): + """V5 verification 5 — this adds a path; it does not touch what is signed. + + The three V5 closes in Perry's own log predate the selection format. They + must keep reading exactly as they did: an evidence file carrying a name and + a date, and an event with no `signoff` key, because there was none. + """ + + SIGNED = { + "TASK-034": "evidence/2026-08/TASK-034-lifecycle.md", + "TASK-103": "evidence/2026-08/TASK-103-design-007-lock.md", + "TASK-047": "evidence/2026-08/TASK-047-dispatch-2026-08-20-1416.md", + } + + def v5_events(self): + log = PERRY_HOME / ".perry" / "events.jsonl" + out = [] + for line in log.read_text().split("\n"): + line = line.strip() + if not line: + continue + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if event.get("rung") == "V5" and event.get("event") == "done": + out.append(event) + return out + + def test_the_three_existing_v5_closes_still_read(self): + events = {e["id"]: e for e in self.v5_events()} + self.assertEqual(set(events), set(self.SIGNED), + "the set of V5 closes in the log moved") + for tid, rel in self.SIGNED.items(): + with self.subTest(tid=tid): + self.assertEqual(events[tid]["evidence"], rel) + self.assertNotIn( + "signoff", events[tid], + "a signature was back-filled onto a close that predates " + "the format — the change adds a path, it does not rewrite " + "history") + + def test_each_signature_document_still_carries_a_name_and_a_date(self): + """What V5 asks for, checked against the files rather than asserted. + These are read here and written nowhere: `perry/` is the PMO's state.""" + state_root = PERRY_HOME / "perry" + for tid, rel in self.SIGNED.items(): + with self.subTest(tid=tid): + text = (state_root / rel).read_text() + self.assertRegex(text, r"\d{4}-\d{2}-\d{2}", + "the signature lost its date") + self.assertTrue( + re.search(r"[Ss]igned off|sign-off|签", text), + "the signature block is no longer findable in the file") + + def test_the_new_record_does_not_claim_to_be_the_old_one(self): + """The old signatures are prose in an evidence file; the new one is a + journal block plus an event. Both are readable, neither is rewritten, + and nothing here converts one into the other.""" + p = Project() + code, out = p.close_v5(p.a_task(), "--checked", "1") + self.assertEqual(code, 0, out) + self.assertIn("signoff", out) + self.assertNotIn("V5 sign-off", (p.root / "BOARD.md").read_text()) + + +class TestV1toV4ClosesAreUntouched(unittest.TestCase): + """Out of scope, asserted rather than assumed: they gain nothing here.""" + + def test_a_v3_close_writes_exactly_what_it_wrote_before(self): + p = Project() + tid = p.a_task() + code, out = p.run("done", tid, "--evidence", "evidence/x.md", + "--rung", "V3") + self.assertEqual(code, 0, out) + self.assertIsNone(out["signoff"]) + self.assertEqual(out["signoff_block"], "") + journal = p.journal() + self.assertNotIn("V5 sign-off", journal) + self.assertNotIn("signed off:", journal) + done = [e for e in p.events() if e["event"] == "done"][0] + self.assertNotIn("signoff", done) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/live_state_expectations.py b/tests/live_state_expectations.py new file mode 100644 index 00000000..14a88d1a --- /dev/null +++ b/tests/live_state_expectations.py @@ -0,0 +1,701 @@ +"""The sweep TASK-113 ran by hand, as a mechanism: a check must not read the +live project as its expected value. + +TASK-113 found five of these in one afternoon and fixed them one at a time; the +pass itself was thrown away, and three more arrived within the week — two the +moment `.perry/config.md` declared its first track (`d90612a`), two more when +PR #14 changed which paths `schema/state-schema.json` says Perry owns. There +was no mechanism, only a memory of having looked. This is the mechanism. + +## What the class IS + +**A check is in this class when a value it read out of the project it lives in +is asserted equal to a literal that enumerates or counts what that project +happens to hold today.** Both halves have to be true: + +1. **It reaches live state.** The value under assertion came from a path + `schema/state-schema.json` declares Perry writes — resolved through the + `State root:` line in `.perry/config.md`, so `BOARD.md` anchored at `state` + means `perry/BOARD.md` here — or from the **parsed payload** of one of + Perry's own tools run with `cwd=` or `--root` pointing into this repository + rather than at a fixture the test built. Nothing about the *names* of those + paths is written down here; the list comes out of the schema, which is why + the move behind instance 8 (a claim added, so `perry/tasks.jsonl` stopped + being unclaimed) changes what this guard considers live rather than being + a fact hard-coded past it. +2. **Its expectation is closed.** The other side of the assertion is a literal + whose value is fixed by the source text: a non-trivial constant, or a + non-empty list/set/tuple/dict *display* — including one reached through a + module or class constant, or through `set(...)`/`sorted(...)` of one. A + display pins cardinality and membership, so the project acquiring or losing + one record falsifies it. `[]`, `{}`, `0`, `1`, `""`, `None` are not closed + in this sense: "nothing is wrong" is a property, quantified over whatever + the project holds, and it is the shape every one of the repairs took. + +The two halves are what keep this from being either useless direction. A test +that reads a fixture it built fails (1) however literal its expectations are — +which is most of the suite, and is the whole point: `test_prioritize` asserts +exact rendered tables against boards it wrote itself, and must keep doing so. A +test that reads BOARD.md and asserts a property *of* what it read — `sum(...) +== len(records)`, `set(report["kinds"]) == {r["kind"] for r in records}`, +`problems == []` — fails (2), because it restates nothing. + +## What it deliberately does NOT catch + +Written down because a guard whose boundary is undocumented gets widened by the +next person until it flags everything. + +- **Containment.** `assertIn("## Why the state root is not `.`", text)` over a + live file is not flagged. Growth cannot falsify it, and `test_md_store` kept + exactly that line through its repair. The cost is real and known: instance 8 + was `assertIn("perry/tasks.jsonl (unclaimed)", …)`, and this guard does not + catch it. Catching it needs a different signal — a *path literal* the + schema's claim list has made stale — and that signal cannot be told apart + from the hundreds of path strings the fixtures legitimately write. +- **A live value used as INPUT.** Instance 2 borrowed `TASK-038` off the live + board and passed it to `perry-task next`; the row closed, the tool answered + "TASK-038 is not a row on the board", and the assertion about flag naming + stopped running. That is the same disease and this guard is blind to it — + the defect is in the fixture, not in an expected value, and every test that + writes a plausible id would look identical. +- **`assertTrue` / `assertFalse` on a live value.** `assertFalse(w[ + "register_declared"])` is genuinely of the class and is not flagged on its + own account: there is no expected side to judge, and the same shape covers + `assertTrue(path.exists())`, which is fine. Instance 7 is still reported, + from the `["main"]` on the line below it. +- **A tool's exit code and its human text.** `assertEqual(proc.returncode, 2)` + and `assertGreater(len(proc.stdout), 40)` are contracts of the TOOL. Only + the JSON payload is a projection of the project, so only it is followed. +- **Code, contracts and templates.** `schema/`, `SKILL.md`, `bin/`, `modes/`, + `state/` and `tests/fixtures/` are not live state. A test that pins an exact + literal against `schema/state-schema.json` SHOULD go red when someone edits + the schema — that is a contract test doing its job, and `test_ownership` is + full of them. The corollary is a known false positive: a payload subtree the + tool builds from its own constants (`perry-task list --json § semantics`) + cannot be told from one it built out of the project. +- **The repository root handed to a helper.** `self.drive("rows", str(ROOT))` + runs a tool on this project and is not reported. Taking `ROOT` as a live + target was tried and reverted: `relative_to(PERRY_HOME)` and the dozen other + bookkeeping uses put 21 fixture-only assertions on the report. +- **Anything outside `tests/`.** `bin/perry-diagnose` reading its own output as + input is the same disease in a different organ (TASK-126), and a different + row. +- **Dynamic reach.** A live read routed through a base class in another module, + or built by string formatting, is invisible here. The analysis is one module + at a time, over the syntax. + +## What it reports today + +Not zero, and the number is the point: `tests/fixtures/live-state-expectations +.json` records every hit with a verdict, so a new one is a red rather than a +line in a report nobody reads. `tests/test_live_state_expectations.py` holds +that baseline to the sweep and reconstructs three of the eight instances out +of history to prove the sweep still finds what it was built for. + +Run: python3 tests/live_state_expectations.py +""" + +from __future__ import annotations + +import argparse +import ast +import fnmatch +import json +import pathlib +import re +import sys +from typing import NamedTuple + +ROOT = pathlib.Path(__file__).resolve().parent.parent + +#: Every hit the sweep makes over this repository, with a verdict on each. +#: Recorded rather than asserted-to-be-empty because it is NOT empty, and a +#: floor nobody wrote down is a floor that drifts upward one silence at a time. +BASELINE = ROOT / "tests" / "fixtures" / "live-state-expectations.json" + +#: Reading a path, as opposed to naming one. `exists`/`stat` are reads too: a +#: test can assert a count of what a live directory holds without opening it. +READ_METHODS = frozenset({ + "read_text", "read_bytes", "open", "read", "readlines", "glob", "rglob", + "iterdir", "exists", "is_file", "is_dir", "stat", +}) + +#: Assertions with an expected side that has to match exactly. `assertIn` and +#: the truth asserts are out on purpose — see the module docstring. +EXACT_ASSERTS = frozenset({ + "assertEqual", "assertNotEqual", "assertListEqual", "assertDictEqual", + "assertSetEqual", "assertTupleEqual", "assertCountEqual", + "assertSequenceEqual", "assertMultiLineEqual", +}) +#: A threshold on a live count is the same defect wearing an inequality — +#: `c9018ae` was `rows_from_store > 20`, made false by one ordinary close. +ORDER_ASSERTS = frozenset({ + "assertGreater", "assertGreaterEqual", "assertLess", "assertLessEqual", +}) + +#: Constructors that carry a literal through unchanged. +PURE_CTORS = frozenset({"set", "frozenset", "list", "tuple", "dict", "sorted"}) + +#: In-place growth, which taints the container the way an assignment would. +MUTATORS = frozenset({"append", "extend", "update", "add", "setdefault"}) + +#: Naming a path, not reading one. +PATH_OPS = frozenset({"str", "os.fspath", "Path", "pathlib.Path"}) + +#: `def` in either flavour. +FUNCTION = (ast.FunctionDef, ast.AsyncFunctionDef) + +SUBPROCESS = frozenset({ + "subprocess.run", "subprocess.check_output", "subprocess.Popen", + "subprocess.call", "subprocess.check_call", +}) + + +# ── which paths are live state ──────────────────────────────────────────── +# Read out of the schema, never listed here. The eight known instances span +# BOARD.md, the journal, the event log, `.perry/config.md` and the claim list +# itself; a guard that named any of them would have missed the others. + +def state_root(root: pathlib.Path) -> str: + """The `State root:` pointer, as a repo-relative prefix (`""` for `.`).""" + config = root / ".perry" / "config.md" + if not config.exists(): + return "" + m = re.search(r"^-\s*State root:\s*(\S+)\s*$", config.read_text(), re.M) + value = (m.group(1) if m else ".").strip("`").strip() + return "" if value == "." else value.strip("/") + + +def live_patterns(root: pathlib.Path = ROOT) -> list[str]: + """Every path the schema declares Perry writes, anchored for this project. + + `anchor: state` hangs the path under the state root; `anchor: project` + hangs it at the repository root. The state root itself is included: a test + that counts what `perry/` holds is reading live state whether or not it + names a file inside it. + """ + schema = json.loads((root / "schema" / "state-schema.json").read_text()) + prefix = state_root(root) + out: set[str] = {".perry"} + if prefix: + out.add(prefix) + declared = list(schema.get("claims", [])) + list(schema.get("files", [])) + for entry in declared: + path = str(entry.get("path", "")).strip("/") + if not path: + continue + if entry.get("anchor") == "state" and prefix: + path = f"{prefix}/{path}" + out.add(path) + return sorted(out) + + +def is_live_path(rel: str, patterns: list[str]) -> bool: + """Does this repo-relative path fall inside the project's own state?""" + if rel is None: + return False + rel = rel.strip("/") + if not rel: + return False + for pat in patterns: + if fnmatch.fnmatch(rel, pat) or fnmatch.fnmatch(rel, pat + "/*"): + return True + # A directory named above a claimed glob — `perry/phase` for + # `perry/phase/[0-9][0-9][0-9]-*.md` — is the same live directory. + if pat.startswith(rel + "/"): + return True + return False + + +# ── findings ────────────────────────────────────────────────────────────── + +class Taint(NamedTuple): + """What a function is holding: values read out of the project, and the + results of tools run against it. The two are separate because a tool's + exit code and its human text are the TOOL's contract — only the parsed + payload is a projection of project state.""" + names: set[str] + tools: set[str] + + +class Finding(NamedTuple): + module: str + lineno: int + test: str + assertion: str + actual: str + expected: str + + @property + def key(self) -> tuple[str, str, str, str, str]: + """Identity WITHOUT the line number, so an edit ten lines above does + not look like a new finding.""" + return (self.module, self.test, self.assertion, self.actual, + self.expected) + + def __str__(self) -> str: + return (f"{self.module}:{self.lineno} {self.test}\n" + f" {self.assertion}(, {self.expected})\n" + f" live: {self.actual}") + + +# ── the syntax the analysis walks ───────────────────────────────────────── + +def _dotted(node: ast.AST) -> str: + """`subprocess.run`, `self.SIGNED`, `T` — or `""` for anything else.""" + if isinstance(node, ast.Name): + return node.id + if isinstance(node, ast.Attribute): + base = _dotted(node.value) + return f"{base}.{node.attr}" if base else "" + return "" + + +def _is_trivial_const(node: ast.AST) -> bool: + return isinstance(node, ast.Constant) and _trivial(node.value) + + +def _trivial(value: object) -> bool: + """A literal that says "nothing", not "exactly this".""" + if value is None or isinstance(value, bool): + return True + if isinstance(value, (int, float)): + return value in (0, 1) + if isinstance(value, (str, bytes)): + return len(value) == 0 + return False + + +class Module: + """One test module, read for the two halves of the class.""" + + def __init__(self, source: str, name: str, patterns: list[str]): + self.name = name + self.tree = ast.parse(source, filename=name) + self.patterns = patterns + #: name → repo-relative path it denotes + self.paths: dict[str, str] = {} + #: name → the literal node it is bound to + self.literals: dict[str, ast.AST] = {} + #: names bound at module level to a live read + self.module_live: set[str] = set() + #: `Class.method` names whose return value is live + self.live_methods: set[str] = set() + #: path names bound inside the function currently being read + self._scope: dict[str, str] = {} + self._collect() + + # -- paths ------------------------------------------------------------- + + def _const_str(self, node: ast.AST) -> str | None: + if isinstance(node, ast.Constant) and isinstance(node.value, str): + return node.value + return None + + def path_of(self, node: ast.AST) -> str | None: + """The repo-relative path this expression denotes, if it is knowable. + + `""` is the repository root, which is a real answer and not a miss — + callers must test `is not None`. + """ + if isinstance(node, ast.Name): + if node.id in self._scope: + return self._scope[node.id] + return self.paths.get(node.id) + if isinstance(node, ast.Attribute): + if node.attr == "parent": + base = self.path_of(node.value) + if not base: # unknown, or the repo root: no parent in here + return None + return base.rsplit("/", 1)[0] if "/" in base else "" + return self.paths.get(_dotted(node)) + if isinstance(node, ast.BinOp) and isinstance(node.op, ast.Div): + base = self.path_of(node.left) + seg = self._const_str(node.right) + if base is None or seg is None: + return None + return f"{base}/{seg}".strip("/") + if isinstance(node, ast.Call): + fn = _dotted(node.func) + if fn in ("str", "os.fspath") and len(node.args) == 1: + return self.path_of(node.args[0]) + if isinstance(node.func, ast.Attribute): + if node.func.attr in ("resolve", "absolute", "expanduser"): + return self.path_of(node.func.value) + if node.func.attr == "joinpath": + base = self.path_of(node.func.value) + segs = [self._const_str(a) for a in node.args] + if base is None or any(s is None for s in segs): + return None + return "/".join([base, *segs]).strip("/") + if fn.endswith("Path") and len(node.args) == 1: + if _dotted(node.args[0]) == "__file__": + return self.name + return self.path_of(node.args[0]) + return None + + def local_paths(self, fn: ast.AST) -> dict[str, str]: + """Path names bound inside one function. + + `log = PERRY_HOME / ".perry" / "events.jsonl"` is where instance 1's + read starts, and it is a local. Two passes so a chain of them settles. + """ + scope: dict[str, str] = {} + saved, self._scope = self._scope, scope + try: + for _ in range(2): + for node in ast.walk(fn): + if isinstance(node, ast.Assign) and len(node.targets) == 1: + name = _dotted(node.targets[0]) + path = self.path_of(node.value) + if name and path is not None: + scope[name] = path + finally: + self._scope = saved + return scope + + # -- collection -------------------------------------------------------- + + def _collect(self) -> None: + for node in self.tree.body: + if isinstance(node, ast.Assign) and len(node.targets) == 1: + self._bind_module(node.targets[0], node.value) + # Two passes: a method that returns live makes its callers live. + for _ in range(2): + for cls in [n for n in self.tree.body + if isinstance(n, ast.ClassDef)]: + for item in cls.body: + if isinstance(item, ast.Assign) and len(item.targets) == 1: + target = _dotted(item.targets[0]) + if target and self._is_literal_node(item.value): + self.literals.setdefault(target, item.value) + self.literals.setdefault(f"self.{target}", + item.value) + if isinstance(item, FUNCTION): + saved = self._scope + self._scope = self.local_paths(item) + try: + if self._returns_live(item, self.taint(item)): + self.live_methods.add(f"self.{item.name}") + finally: + self._scope = saved + for node in self.tree.body: + if isinstance(node, ast.Assign) and len(node.targets) == 1: + if self.is_live(node.value, Taint(set(), set())): + self.module_live |= _bound_names(node.targets[0]) + + def _bind_module(self, target: ast.AST, value: ast.AST) -> None: + name = _dotted(target) + if not name: + return + path = self.path_of(value) + if path is not None: + self.paths[name] = path + if self._is_literal_node(value): + self.literals[name] = value + + def _is_literal_node(self, node: ast.AST) -> bool: + if isinstance(node, ast.Constant): + return True + if isinstance(node, (ast.List, ast.Set, ast.Tuple)): + return all(self._is_literal_node(e) for e in node.elts) + if isinstance(node, ast.Dict): + return all(k is not None and self._is_literal_node(k) + for k in node.keys) + return False + + def _returns_live(self, fn: ast.AST, taint: Taint) -> bool: + return any(isinstance(n, ast.Return) and n.value is not None + and self.is_live(n.value, taint) + for n in ast.walk(fn)) + + # -- half one: does this expression reach live state? ------------------ + + def live_source(self, node: ast.AST, + tools: frozenset[str] = frozenset()) -> bool: + """This node, on its own, reads the project living around the test.""" + if not isinstance(node, ast.Call): + return False + fn = _dotted(node.func) + if fn in ("open", "io.open") and node.args: + return is_live_path(self.path_of(node.args[0]), self.patterns) + if isinstance(node.func, ast.Attribute) \ + and node.func.attr in READ_METHODS: + return is_live_path(self.path_of(node.func.value), self.patterns) + if fn in ("json.loads", "json.load") and node.args: + return self._reaches_tool(node.args[0], tools) + if fn in SUBPROCESS: + return False # the run itself; only its PAYLOAD is the project + if fn in PATH_OPS or self.path_of(node) is not None: + return False # naming a path is not reading one + # Any other call handed a live path reads it — the callee is a helper + # in this module or a sibling, and `assert_round_trips(M.CONFIG, path)` + # is how instance 6 reached `.perry/config.md`. The repository ROOT + # itself does not count — `relative_to(PERRY_HOME)` and `str(ROOT)` + # handed to a helper are overwhelmingly bookkeeping, and taking them + # as reads put 21 fixture-only assertions on the report. + return any(is_live_path(self.path_of(a), self.patterns) + for a in _call_operands(node)) + + def _reaches_tool(self, node: ast.AST, tools: frozenset[str]) -> bool: + """Does this expression carry the output of a tool run on this repo?""" + for n in ast.walk(node): + if isinstance(n, ast.Call) and _dotted(n.func) in SUBPROCESS \ + and self._tool_reads_this_project(n): + return True + if isinstance(n, (ast.Name, ast.Attribute)) \ + and _dotted(n) in tools: + return True + return False + + def _tool_reads_this_project(self, call: ast.Call) -> bool: + """A Perry tool pointed at this repository, not at a fixture. + + A test says which project it means in one of three places, read in + this order: `--root `, `cwd=`, or a state path among the + arguments. **With none of them the answer is no** — the tool would in + fact inherit the runner's cwd and so read this repository, but + `--help` and `--version` runs are the bulk of that population and none + of them touches state. A stated blind spot, not a claim: say + `cwd=ROOT` and the guard sees you. + """ + operands = _call_operands(call) + for i, node in enumerate(operands): + if isinstance(node, ast.Constant) and node.value == "--root": + nxt = operands[i + 1] if i + 1 < len(operands) else None + return nxt is not None and self.path_of(nxt) is not None + kwargs = {kw.arg: kw.value for kw in call.keywords if kw.arg} + if "cwd" in kwargs: + return self.path_of(kwargs["cwd"]) is not None + return any(is_live_path(self.path_of(a), self.patterns) + for a in operands) + + def is_live(self, node: ast.AST, taint: Taint) -> bool: + for n in ast.walk(node): + if self.live_source(n, taint.tools): + return True + if isinstance(n, (ast.Name, ast.Attribute)): + if _dotted(n) in taint.names: + return True + if isinstance(n, ast.Call) \ + and _dotted(n.func) in self.live_methods: + return True + return False + + def taint(self, fn: ast.AST) -> Taint: + """Names inside one function that hold something read out of the + project (`names`), and names holding a tool run against it (`tools`). + + Three passes rather than a real fixpoint: the deepest chain in this + suite is four assignments and the analysis is advisory. + """ + taint = Taint(set(self.module_live), set()) + for _ in range(3): + before = len(taint.names) + len(taint.tools) + for node in ast.walk(fn): + bound: set[str] = set() + value: ast.AST | None = None + if isinstance(node, ast.Assign): + value = node.value + for t in node.targets: + bound |= _bound_names(t) + elif isinstance(node, (ast.AnnAssign, ast.AugAssign)): + value, bound = node.value, _bound_names(node.target) + elif isinstance(node, ast.For): + value, bound = node.iter, _bound_names(node.target) + elif isinstance(node, ast.With): + for item in node.items: + if item.optional_vars is not None \ + and self.is_live(item.context_expr, taint): + taint.names.update( + _bound_names(item.optional_vars)) + elif isinstance(node, ast.Call) \ + and isinstance(node.func, ast.Attribute) \ + and node.func.attr in MUTATORS: + if any(self.is_live(a, taint) for a in node.args): + taint.names.update(_bound_names(node.func.value)) + if value is None or not bound: + continue + if self.is_live(value, taint): + taint.names.update(bound) + if self._reaches_tool(value, frozenset(taint.tools)): + taint.tools.update(bound) + if len(taint.names) + len(taint.tools) == before: + break + return taint + + # -- half two: is the expectation closed? ------------------------------ + + def closed_literal(self, node: ast.AST, depth: int = 0) -> bool: + if depth > 4: + return False + if isinstance(node, (ast.Name, ast.Attribute)): + bound = self.literals.get(_dotted(node)) + return bound is not None and self.closed_literal(bound, depth + 1) + if isinstance(node, ast.Constant): + return not _trivial(node.value) + if isinstance(node, (ast.List, ast.Set, ast.Tuple)): + # Every element fixed by the source, or it is not an enumeration: + # `(expected["open"], expected["closed"])` restates nothing. + return bool(node.elts) and all( + self.closed_literal(e, depth + 1) or _is_trivial_const(e) + for e in node.elts) + if isinstance(node, ast.Dict): + # The KEYS are what a dict display pins: `{"setting": n}` + # says "one kind, named `setting`" whatever the count beside it is. + return bool(node.keys) and all( + k is not None and isinstance(k, ast.Constant) + for k in node.keys) + if isinstance(node, ast.Call) and len(node.args) == 1 \ + and _dotted(node.func) in PURE_CTORS: + return self.closed_literal(node.args[0], depth + 1) + return False + + # -- the sweep --------------------------------------------------------- + + def findings(self) -> list[Finding]: + out: list[Finding] = [] + for cls in ast.walk(self.tree): + if not isinstance(cls, ast.ClassDef): + continue + for fn in cls.body: + if not isinstance(fn, FUNCTION): + continue + out.extend(self._findings_in(fn, f"{cls.name}.{fn.name}")) + return sorted(out, key=lambda f: (f.module, f.lineno)) + + def _findings_in(self, fn: ast.AST, where: str) -> list[Finding]: + saved, self._scope = self._scope, self.local_paths(fn) + try: + return self._scan(fn, where) + finally: + self._scope = saved + + def _scan(self, fn: ast.AST, where: str) -> list[Finding]: + taint = self.taint(fn) + out: list[Finding] = [] + for node in ast.walk(fn): + if not isinstance(node, ast.Call) \ + or not isinstance(node.func, ast.Attribute): + continue + name = node.func.attr + if name not in EXACT_ASSERTS and name not in ORDER_ASSERTS: + continue + if len(node.args) < 2: + continue + left, right = node.args[0], node.args[1] + for live, lit in ((left, right), (right, left)): + if self.closed_literal(lit) and self.is_live(live, taint): + out.append(Finding( + module=self.name, lineno=node.lineno, test=where, + assertion=name, + actual=_clip(ast.unparse(live)), + expected=_clip(ast.unparse(lit)))) + break + return out + + +def _call_operands(call: ast.Call) -> list[ast.AST]: + """Positional arguments, flattened through the one list a command is + usually spelled as, so `--root` and its value stay adjacent.""" + out: list[ast.AST] = [] + for arg in call.args: + if isinstance(arg, (ast.List, ast.Tuple)): + out.extend(arg.elts) + elif isinstance(arg, ast.Starred): + out.append(arg.value) + else: + out.append(arg) + out.extend(kw.value for kw in call.keywords if kw.arg not in ("cwd",)) + return out + + +def _bound_names(target: ast.AST) -> set[str]: + if isinstance(target, ast.Name): + return {target.id} + if isinstance(target, (ast.Tuple, ast.List)): + return set().union(*(_bound_names(e) for e in target.elts)) \ + if target.elts else set() + if isinstance(target, ast.Attribute): + return {_dotted(target)} - {""} + if isinstance(target, ast.Subscript): + return _bound_names(target.value) + if isinstance(target, ast.Starred): + return _bound_names(target.value) + return set() + + +def _clip(text: str, width: int = 96) -> str: + text = " ".join(text.split()) + return text if len(text) <= width else text[:width - 1] + "…" + + +# ── entry points ────────────────────────────────────────────────────────── + +def scan_source(source: str, name: str, + root: pathlib.Path = ROOT) -> list[Finding]: + """Every finding in one module's source text, named however you like. + + Takes source rather than a path so a historical revision — `git show + :tests/` — can be swept without being written back into + the tree it was taken from. + """ + return Module(source, name, live_patterns(root)).findings() + + +def sweep(root: pathlib.Path = ROOT) -> list[Finding]: + """The whole suite as it stands.""" + patterns = live_patterns(root) + out: list[Finding] = [] + for path in sorted((root / "tests").glob("test_*.py")): + rel = path.relative_to(root).as_posix() + out.extend(Module(path.read_text(), rel, patterns).findings()) + return out + + +def recorded() -> dict[tuple[str, str, str, str, str], dict]: + """The baseline, keyed the way a finding is.""" + if not BASELINE.exists(): + return {} + entries = json.loads(BASELINE.read_text())["findings"] + return {(e["module"], e["test"], e["assertion"], e["actual"], + e["expected"]): e for e in entries} + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + ap.add_argument("--json", action="store_true", help="machine-readable") + ap.add_argument("--root", default="", help="the repo to sweep") + ap.add_argument("--record", action="store_true", + help="rewrite the baseline, KEEPING every verdict already " + "written for a finding that is still there") + args = ap.parse_args(argv) + root = pathlib.Path(args.root).resolve() if args.root else ROOT + found = sweep(root) + if args.record: + known = recorded() + BASELINE.write_text(json.dumps({ + "note": "Every hit of tests/live_state_expectations.py over this " + "repository. A finding with no verdict has not been " + "looked at; `instance` means a row is owed for it.", + "findings": [ + {**f._asdict(), + "verdict": known.get(f.key, {}).get("verdict", ""), + "why": known.get(f.key, {}).get("why", "")} + for f in found], + }, indent=2) + "\n") + print(f"recorded {len(found)} finding(s) in " + f"{BASELINE.relative_to(ROOT)}") + return 0 + if args.json: + print(json.dumps([f._asdict() for f in found], indent=2)) + else: + print("\n".join(str(f) for f in found) if found + else "no check reads live project state as its expected value") + print(f"\n{len(found)} finding(s) · live paths: " + f"{len(live_patterns(root))} declared in schema") + return 1 if found else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_live_state_expectations.py b/tests/test_live_state_expectations.py new file mode 100644 index 00000000..708c77ca --- /dev/null +++ b/tests/test_live_state_expectations.py @@ -0,0 +1,419 @@ +"""The guard over the guard: `tests/live_state_expectations.py` still finds +checks that read the live project as their expected value. + +`tests/live_state_expectations.py` carries the definition of the class and the +list of what it deliberately does not catch. This module is the evidence that +the definition still bites, and it is built the only way that proves anything: +**the instances come out of git, not out of my hands.** A guard tested against +examples written the same afternoon as the guard proves that the author can +restate a regex twice. + +Three of the eight known instances are checked in verbatim under +`tests/fixtures/live-state/`, each a whole module as it stood in the commit +before its repair: + +| fixture | commit | repaired by | the assertion | +|---|---|---|---| +| `md_store.before.py` | `d90612a` | `f3c4461` | `report["kinds"] == {"setting": len(records)}` | +| `track_attribution.before.py` | `d90612a` | `f3c4461` | `[t["track"] …] == ["main"]` | +| `v5_signoff.before.py` | `e116f8a` | `cbbc41a` | `set(events) == set(self.SIGNED)` | + +`d90612a` is `f3c4461^` and `e116f8a` is `cbbc41a^`: each fixture is the module +as the repair found it. + +Whole modules rather than excerpts, because trimming to the interesting class +is hand-writing an approximation by another name — the analysis reads +module-level bindings, class attributes and sibling methods, and an excerpt +would be a different program. `test_the_fixtures_are_what_git_holds` compares +them byte-for-byte against `git show` when the history is reachable, and +`test_the_fixtures_have_not_been_edited` pins their SHA-256 for the CI checkout +that is shallow and cannot. + +**Verification 2 is the other half and is not decoration.** A guard that still +flags the repaired form is measuring the wrong thing, so the repaired versions +are checked too — for `test_track_attribution` and `test_v5_signoff` that is +the file in the tree today, byte-identical to its fix commit; for +`test_md_store` the repaired assertion is gone and three others in that module +are not, which the baseline records with a verdict apiece. + +**The floor is a recorded number, not zero.** Six hits over 65 modules, three +of them real. Asserting zero would have meant either widening three verdicts +into silence or fixing three rows this one is explicitly not allowed to fix. + +Run: python3 tests/parallel test_live_state_expectations +""" + +from __future__ import annotations + +import ast +import hashlib +import json +import pathlib +import subprocess +import sys +import tempfile +import unittest + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) + +import live_state_expectations as L # noqa: E402 + +ROOT = L.ROOT +FIXTURES = ROOT / "tests" / "fixtures" / "live-state" + + +class Reconstruction(unittest.TestCase): + """One instance, as it stood before its repair and after it.""" + + fixture = "" + #: `git show :` — what the fixture is a copy of. + commit = "" + path = "" + sha256 = "" + #: The repair, and the file that carries it today (`""` when the module has + #: moved on since, and only the assertion's absence can be checked). + repaired_by = "" + repaired_file = "" + #: What the guard must say about the unrepaired module, and what must have + #: stopped being said about the repaired one. + flagged_test = "" + flagged_expected = "" + + def before(self) -> list[L.Finding]: + return L.scan_source((FIXTURES / self.fixture).read_text(), self.path) + + def hits(self, findings) -> list[L.Finding]: + return [f for f in findings + if f.test == self.flagged_test + and f.expected == self.flagged_expected] + + +class Instance6(Reconstruction): + """`.perry/config.md` declared one track and every config record stopped + being a `setting`. The literal is a dict DISPLAY whose value is computed — + `{"setting": len(records)}` — so a guard that only looked for wholly + constant expectations would have walked past it: what the line pins is the + KEY SET, one kind named `setting`, whatever the count beside it says.""" + + fixture = "md_store.before.py" + commit = "d90612a149e44b6e76523df04749308bc9b0d201" + path = "tests/test_md_store.py" + sha256 = ("7f87dbc64c6c3004f90a023bd6eee0669b344e6e0409ee44cb42548" + "fe3b477d7") + repaired_by = "f3c44617f2d32abf102de35eeda0ea7e33eee2a0" + repaired_file = "tests/test_md_store.py" + flagged_test = ("TestThisRepositoryIsReproducedByteForByte." + "test_config_including_its_prose_section") + flagged_expected = "{'setting': len(records)}" + + def test_the_unrepaired_module_is_flagged(self): + self.assertEqual(len(self.hits(self.before())), 1, + "\n".join(str(f) for f in self.before())) + + def test_the_repair_is_not_flagged(self): + current = (ROOT / self.repaired_file) + found = L.scan_source(current.read_text(), self.repaired_file) + self.assertEqual(self.hits(found), []) + + def test_the_derived_form_the_repair_used_is_not_flagged(self): + """The shape the repair replaced it with, on its own. + + `set(report["kinds"]) == {r["kind"] for r in records}` reads the same + live file and restates nothing about it, which is exactly the + difference the class turns on. If this ever flags, the guard has + stopped measuring closedness and started measuring "touched a file".""" + found = L.scan_source(REPAIRED_SHAPE, "tests/test_x.py") + self.assertEqual([str(f) for f in found], []) + + +class Instance7(Reconstruction): + """The same declaration reddened `test_track_attribution`, which asserted + that Perry itself has no track register. The repair proves the no-op + property on a project that HAS no register instead of on this one.""" + + fixture = "track_attribution.before.py" + commit = "d90612a149e44b6e76523df04749308bc9b0d201" + path = "tests/test_track_attribution.py" + sha256 = ("7bbc7cad5fe9ee9e5b081e896c7e0126eec71e4c72fc72b115259d4" + "87e5320d4") + repaired_by = "f3c44617f2d32abf102de35eeda0ea7e33eee2a0" + repaired_file = "tests/test_track_attribution.py" + flagged_test = ("TestPerrysOwnProjectIsUnmoved." + "test_it_still_reads_one_project_track") + flagged_expected = "['main']" + + def test_the_unrepaired_module_is_flagged(self): + self.assertEqual(len(self.hits(self.before())), 1, + "\n".join(str(f) for f in self.before())) + + def test_the_repaired_module_is_clean_end_to_end(self): + """Not just the one line: the whole module, which is what item 2 of + the verification asks for. The repaired class still asserts `["main"]` + — against a project it built itself, where a literal is a fixture and + not a snapshot.""" + text = (ROOT / self.repaired_file).read_text() + self.assertIn('["main"]', text, "the literal did not survive the " + "repair, so this proves nothing") + self.assertEqual(L.scan_source(text, self.repaired_file), []) + + +class Instance1(Reconstruction): + """TASK-113's: `test_v5_signoff` named exactly three V5 closes and three + more were signed the day it shipped. The literal is a CLASS ATTRIBUTE + behind a `set()` call — `set(self.SIGNED)` — which is why the analysis + folds constants through both.""" + + fixture = "v5_signoff.before.py" + commit = "e116f8a288a2f0d159c1d3bc03b9ce9eb44c32af" + path = "tests/test_v5_signoff.py" + sha256 = ("89eb77fbc216d1363915f4f9cabe9405d6c15e6b51663da65a3cf0c" + "19284700a") + repaired_by = "cbbc41af5e81f9b552ba8797eb17727d7e1934f0" + repaired_file = "tests/test_v5_signoff.py" + flagged_test = ("TestHistoryIsNotRewritten." + "test_the_three_existing_v5_closes_still_read") + flagged_expected = "set(self.SIGNED)" + + def test_the_unrepaired_module_is_flagged(self): + self.assertEqual(len(self.hits(self.before())), 1, + "\n".join(str(f) for f in self.before())) + + def test_the_repaired_module_is_clean_end_to_end(self): + text = (ROOT / self.repaired_file).read_text() + self.assertEqual(L.scan_source(text, self.repaired_file), []) + + def test_the_reader_the_repair_replaced_it_with_is_not_flagged(self): + """`assertEqual(problems, [])` over whatever the journal holds. The + empty display is what separates a property from a snapshot, and it is + the shape all five of TASK-113's repairs converged on.""" + found = L.scan_source(QUANTIFIED_SHAPE, "tests/test_x.py") + self.assertEqual([str(f) for f in found], []) + + +#: The repaired `test_md_store` assertion, standing alone. Reads the same live +#: file; expects nothing the file does not say. +REPAIRED_SHAPE = ''' +import pathlib, unittest +ROOT = pathlib.Path(__file__).resolve().parent.parent + +class T(unittest.TestCase): + def test_it(self): + records, report = derive(ROOT / ".perry" / "config.md") + self.assertEqual(sum(report["kinds"].values()), len(records)) + self.assertEqual(set(report["kinds"]), {r["kind"] for r in records}) + self.assertEqual("track" in report["kinds"], has_register) +''' + +#: The repaired `test_v5_signoff` shape: quantified over the log, expecting +#: emptiness rather than a roster. +QUANTIFIED_SHAPE = ''' +import pathlib, unittest +ROOT = pathlib.Path(__file__).resolve().parent.parent + +class T(unittest.TestCase): + def test_it(self): + log = ROOT / ".perry" / "events.jsonl" + problems = [p for line in log.read_text().split("\\n") + for p in unreadable(line)] + self.assertEqual(problems, []) + self.assertEqual(len(problems), 0) +''' + +#: A fixture-built project asserting exact literals — most of this suite, and +#: the thing a too-broad guard would drown in. +FIXTURE_SHAPE = ''' +import pathlib, tempfile, unittest +ROOT = pathlib.Path(__file__).resolve().parent.parent + +class T(unittest.TestCase): + def test_it(self): + root = pathlib.Path(tempfile.mkdtemp()) + (root / "BOARD.md").write_text(BOARD) + rows = read(root / "BOARD.md") + self.assertEqual([r["id"] for r in rows], ["TASK-001", "TASK-002"]) + self.assertEqual(rows[0]["priority"], "P1") +''' + +#: The class itself, in one method, reached three different ways. +LIVE_SHAPE = ''' +import json, pathlib, subprocess, sys, unittest +ROOT = pathlib.Path(__file__).resolve().parent.parent +TOOL = ROOT / "bin" / "perry-state" + +class T(unittest.TestCase): + def test_a_file(self): + rows = read((ROOT / "perry" / "BOARD.md").read_text()) + self.assertEqual([r["id"] for r in rows], ["TASK-001"]) + + def test_a_count(self): + rows = read((ROOT / "perry" / "tasks.jsonl").read_text()) + self.assertGreater(len(rows), 40) + + def test_a_tool(self): + proc = subprocess.run([sys.executable, str(TOOL), "--json"], + capture_output=True, text=True, cwd=ROOT) + payload = json.loads(proc.stdout) + self.assertEqual(payload["tracks"], ["main"]) +''' + + +class TestTheFixturesAreHistoryAndNotMyHandwriting(unittest.TestCase): + """Verification 1 says *reconstruct*, and a fixture nobody can audit is + indistinguishable from an approximation typed to match the guard.""" + + CASES = (Instance6, Instance7, Instance1) + + def test_the_fixtures_have_not_been_edited(self): + """Runs everywhere, including the shallow CI checkout.""" + for case in self.CASES: + with self.subTest(fixture=case.fixture): + blob = (FIXTURES / case.fixture).read_bytes() + self.assertEqual(hashlib.sha256(blob).hexdigest(), + case.sha256) + + def test_the_fixtures_are_what_git_holds(self): + """Runs where the history is reachable. `.github/workflows/ci.yml` + uses `actions/checkout@v4` at its default depth of 1, so this SKIPS on + CI — which is why the digest above exists and is not redundant.""" + for case in self.CASES: + with self.subTest(fixture=case.fixture): + proc = subprocess.run( + ["git", "show", f"{case.commit}:{case.path}"], + capture_output=True, cwd=ROOT) + if proc.returncode != 0: + self.skipTest(f"{case.commit[:7]} is not in this checkout: " + f"{proc.stderr.decode()[:120]}") + self.assertEqual((FIXTURES / case.fixture).read_bytes(), + proc.stdout, + f"{case.fixture} is not byte-identical to " + f"{case.commit[:7]}:{case.path}") + + def test_every_fixture_still_parses_as_python(self): + for case in self.CASES: + with self.subTest(fixture=case.fixture): + ast.parse((FIXTURES / case.fixture).read_text()) + + +class TestTheTwoHalvesAreBothLoadBearing(unittest.TestCase): + """Anti-vacuity. Each half is removed in turn and the answer must move. + + A guard nobody has watched go red is not a guard, and a guard nobody has + watched go GREEN on the honest form is worse — it teaches people to route + around it. + """ + + def scan(self, source: str) -> list[L.Finding]: + return L.scan_source(source, "tests/test_x.py") + + def test_a_fixture_built_project_is_never_flagged(self): + self.assertEqual(self.scan(FIXTURE_SHAPE), []) + + def test_live_state_with_a_closed_expectation_is_flagged_three_ways(self): + found = {f.test.split(".")[-1] for f in self.scan(LIVE_SHAPE)} + self.assertEqual(found, {"test_a_file", "test_a_count", "test_a_tool"}, + "\n".join(str(f) for f in self.scan(LIVE_SHAPE))) + + def test_an_empty_expectation_over_live_state_is_not_flagged(self): + self.assertEqual(self.scan(QUANTIFIED_SHAPE), []) + + def test_a_derived_expectation_over_live_state_is_not_flagged(self): + self.assertEqual(self.scan(REPAIRED_SHAPE), []) + + def test_moving_the_read_off_live_state_clears_every_finding(self): + """Half one, removed: the same assertions against `schema/` — a + contract, which a test SHOULD pin exactly.""" + contract = LIVE_SHAPE.replace('"perry" / "BOARD.md"', + '"schema" / "state-schema.json"') + contract = contract.replace('"perry" / "tasks.jsonl"', + '"schema" / "contract-shapes.json"') + contract = contract.replace("cwd=ROOT", "cwd=self.tmp") + self.assertEqual(self.scan(contract), []) + + +class TestTheLiveSetIsReadOutOfTheSchema(unittest.TestCase): + """Instance 8's lesson: the literals that went stale were about *which + paths the schema declares Perry owns*. A guard holding its own list of + state files would have been wrong the day PR #14 merged.""" + + def test_this_repository(self): + patterns = L.live_patterns(ROOT) + for live in ("perry/BOARD.md", "perry/tasks.jsonl", "perry/OKR.md", + "perry/journal/2026-08/2026-08-20.md", + ".perry/events.jsonl", ".perry/config.md"): + with self.subTest(path=live): + self.assertTrue(L.is_live_path(live, patterns)) + for code in ("schema/state-schema.json", "SKILL.md", + "bin/perry-task", "tests/fixtures/sample-project/BOARD.md", + "modes/queue.md", "state/BOARD_TEMPLATE.md"): + with self.subTest(path=code): + self.assertFalse(L.is_live_path(code, patterns)) + + def project(self, state_root: str) -> list[str]: + """A two-claim project with the `State root:` this asks for.""" + tmp = tempfile.TemporaryDirectory() + self.addCleanup(tmp.cleanup) + root = pathlib.Path(tmp.name) + (root / ".perry").mkdir() + (root / ".perry" / "config.md").write_text( + f"# Perry configuration\n\n- State root: {state_root}\n") + (root / "schema").mkdir() + (root / "schema" / "state-schema.json").write_text(json.dumps({ + "claims": [{"path": "BOARD.md", "anchor": "state"}, + {"path": ".perry/", "anchor": "project"}], + "files": [], + })) + return L.live_patterns(root) + + def test_the_same_claim_lands_where_the_state_root_points(self): + """The proof that it is derived rather than listed. One schema, one + claim, two projects: `BOARD.md` is at the top of the first and inside + `docs/` in the second, and neither is written down here.""" + flat = self.project(".") + self.assertTrue(L.is_live_path("BOARD.md", flat)) + self.assertFalse(L.is_live_path("docs/BOARD.md", flat)) + + nested = self.project("docs") + self.assertTrue(L.is_live_path("docs/BOARD.md", nested)) + self.assertFalse(L.is_live_path("BOARD.md", nested), + "the claim followed the state root, so a file at the " + "top is no longer the one the schema declares") + for both in (flat, nested): + self.assertTrue(L.is_live_path(".perry/events.jsonl", both), + "`anchor: project` never moves") + + +class TestTheFloorIsRecordedNotAssumed(unittest.TestCase): + """Verification 3. Every hit is in the baseline with a verdict; a hit that + is not is a red, and so is a baseline entry the sweep no longer makes.""" + + def setUp(self): + self.found = L.sweep(ROOT) + self.known = L.recorded() + + def test_the_baseline_and_the_sweep_agree(self): + new = [str(f) for f in self.found if f.key not in self.known] + gone = [k for k in self.known if k not in {f.key for f in self.found}] + self.assertEqual(new, [], "a check has started reading live project " + "state as its expected value") + self.assertEqual(gone, [], "a recorded finding is gone — re-record " + "with `--record`, which keeps the verdicts") + + def test_every_recorded_finding_carries_a_verdict_and_a_reason(self): + for entry in json.loads(L.BASELINE.read_text())["findings"]: + with self.subTest(where=f"{entry['module']}:{entry['lineno']}"): + self.assertIn(entry["verdict"], ("instance", "false positive")) + self.assertGreater(len(entry["why"]), 60, + "a verdict without a reason is a silence") + + def test_the_floor_is_not_claimed_to_be_zero(self): + """The number, stated. Three of the six are real and each owes a row; + this module ships the mechanism and fixes none of them.""" + verdicts = [e["verdict"] + for e in json.loads(L.BASELINE.read_text())["findings"]] + self.assertEqual(len(self.found), len(verdicts)) + self.assertEqual(verdicts.count("instance"), 3) + self.assertEqual(verdicts.count("false positive"), 3) + + +if __name__ == "__main__": + unittest.main()