From 07399e700489b6f539ebf11b53cd8edea185f721 Mon Sep 17 00:00:00 2001 From: melbinjp Date: Wed, 19 Aug 2026 14:58:49 +0530 Subject: [PATCH] The verdict says how much of the documentation it actually covered Measured, not imagined. Across nine public repositories docproof read 972 of 3,782 documentation files under the default scope - 25.7%. Re-running two of them over the whole tree took langwatch from 1 broken to 19 and cherry-studio from 23 to 111. Those runs were never clean; they were narrow. The information was already printed, in the header. The verdict was forty lines below it and said 'Nothing contradicted. 149 claims checked' on its own, which is the exact thing the README promises cannot happen: 'a clean report over two files in a project with three hundred cannot be mistaken for a clean report over three hundred.' Printed forty lines apart it can be, and the line anyone quotes from a CI log is the last one. So render() now takes read/unread and appends the coverage to the verdict itself - to the BROKEN verdict too, because '19 broken' over a fifth of a tree misleads exactly as much as 'nothing contradicted' over a fifth. Full coverage adds no sentence, because a reassurance on every clean run is the noise that teaches people to skip the block. Three tests. The second one failed first time because I built a fixture with no deleted path and asserted it was broken - the make_repo docstring is emphatic that a later removal is the whole rule, and I had ignored it. Fixed to use documented_before plus deleted, which is real drift rather than an illustration. --- README.md | 16 ++++++++++++ src/docproof/cli.py | 5 ++-- src/docproof/report.py | 25 ++++++++++++++++++- tests/test_coverage.py | 56 ++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 99 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index c76e129..369929a 100644 --- a/README.md +++ b/README.md @@ -128,6 +128,22 @@ documentation files it did not read and which directories they are in, so a clea over two files in a project with three hundred cannot be mistaken for a clean report over three hundred. See [what gets read](#what-gets-read). +**That sentence was printed in the header and the verdict was at the bottom, which is not the +same promise.** A run would end `Nothing contradicted. 149 claims checked` with the coverage +note forty lines above it, and the line anyone quotes from a CI log is the last one. So the +verdict now carries it: + +``` +1 broken, 149 checked, 497 not judged. +This judged 11 of 383 documentation file(s). 372 were never read, so this verdict +covers 2% of the documentation in this project. +``` + +That is a real run against a real repository. It was measured: over nine public repositories +docproof read **972 of 3,782** documentation files under the default scope, and re-running two +of them across the whole tree took one from 1 finding to 19 and another from 23 to 111. Those +runs were never clean; they were narrow, and only the header said so. + ## What it checks Two things, each done properly: diff --git a/src/docproof/cli.py b/src/docproof/cli.py index 56c7e6a..f972005 100644 --- a/src/docproof/cli.py +++ b/src/docproof/cli.py @@ -292,7 +292,8 @@ def main(argv: Sequence[str] | None = None) -> int: outcomes.append(outcome) report = Report(project=project, outcomes=outcomes) print(f"docproof {__version__} — {project.root.name}, {len(documents)} document(s)") - report_coverage(project, unread_documents(project.root, in_scope, tracked)) + unread = unread_documents(project.root, in_scope, tracked) + report_coverage(project, unread) report_set_aside(historical, disclaimed) # Same principle, one level down, and it applies harder: nobody asked for this rule. # A `Before:` label is the tool deciding by itself that a block is not a claim, so the @@ -306,7 +307,7 @@ def main(argv: Sequence[str] | None = None) -> int: if superseded: print(f" labelled superseded by the prose above, not judged: {', '.join(sorted(superseded))}") print() - print(report.render(show_skips=args.show_skips)) + print(report.render(show_skips=args.show_skips, read=len(documents), unread=len(unread))) if args.exit_zero and not report.stopped_checking: return 0 return report.exit_code diff --git a/src/docproof/report.py b/src/docproof/report.py index cb8b76f..7c04aeb 100644 --- a/src/docproof/report.py +++ b/src/docproof/report.py @@ -79,7 +79,21 @@ def _silence_line(self, outcome: Outcome) -> str: marker = WARN if verdict.alarming or verdict.kind is Silence.UNKNOWN else DASH return f"{marker} {outcome.verifier}: found nothing to check — {verdict.detail}." - def render(self, *, show_skips: bool = False) -> str: + def render(self, *, show_skips: bool = False, read: int = 0, unread: int = 0) -> str: + """`read` and `unread` put the document-level coverage INTO the verdict. + + **Measured defect, 2026-08-19.** The coverage note prints in the header and the verdict + prints at the bottom, so the last line of a run said `Nothing contradicted. 28 claims + checked` while 18 of that project's 26 documentation files had never been opened. The + README promises the opposite in as many words: *"a clean report over two files in a + project with three hundred cannot be mistaken for a clean report over three hundred."* + Printed forty lines apart, it can be, and the line people quote is the last one. + + The measurement that forced this: across nine real repositories docproof read **972 of + 3,782** documentation files, 25.7%. Re-run over the whole tree, langwatch went from 1 + broken to 19 and cherry-studio from 23 to 111. Those runs were not clean, they were + narrow, and only the header said so. + """ lines: list[str] = [] root = self.project.root @@ -129,4 +143,13 @@ def render(self, *, show_skips: bool = False) -> str: lines.append(f"Nothing contradicted. {self.checked} claims checked, {self.skipped} not judged.") if self.skipped and not show_skips: lines.append("Run with --show-skips to see what was left unjudged and why.") + # Attached to the verdict itself, and to the BROKEN verdict too: "19 broken" over a + # fifth of the tree is as easy to misread as "nothing contradicted" over a fifth. + if unread: + total = read + unread + lines.append( + f"This judged {read} of {total} documentation file(s). " + f"{unread} were never read, so this verdict covers " + f"{100 * read // total}% of the documentation in this project." + ) return "\n".join(lines) diff --git a/tests/test_coverage.py b/tests/test_coverage.py index f1ba4b9..1e56e77 100644 --- a/tests/test_coverage.py +++ b/tests/test_coverage.py @@ -149,3 +149,59 @@ def test_the_historical_list_is_capped_and_says_the_remainder(make_repo: Callabl # The whole block stays on one line, so its length is the thing that was wrong before. line = next(ln for ln in out.splitlines() if "describing the past" in ln) assert len(line) < 500, line[:200] + + +def test_the_verdict_itself_says_how_much_it_covered(make_repo: Callable[..., Path], capsys) -> None: + """**The coverage note was in the header and the verdict was at the bottom.** + + A run ended `Nothing contradicted. 28 claims checked` while 18 of that project's 26 + documentation files had never been opened. The README promises the opposite in as many + words: *"a clean report over two files in a project with three hundred cannot be mistaken + for a clean report over three hundred."* Forty lines apart, it can be, and the line people + quote from a CI log is the last one. + + Measured before it was changed: across nine real repositories docproof read **972 of 3,782** + documentation files, 25.7 per cent. Re-run over the whole tree, langwatch went from 1 broken + to 19 and cherry-studio from 23 to 111. Those runs were not clean; they were narrow. + """ + files = {"README.md": README, "src/thing.py": "x = 1\n"} + for n in range(9): + files[f"elsewhere/note{n}.md"] = "# Note\n" + repo = make_repo(files) + main([str(repo)]) + out = capsys.readouterr().out + assert "Nothing contradicted." in out + verdict = out.strip().splitlines()[-1] + assert "9 were never read" in verdict + assert "10% of the documentation" in verdict + + +def test_a_broken_verdict_carries_the_coverage_too(make_repo: Callable[..., Path], capsys) -> None: + """ "19 broken" over a fifth of the tree is as easy to misread as "nothing contradicted" + over a fifth, so the sentence is attached to both verdicts rather than only the clean one.""" + files = {"src/thing.py": "x = 1\n"} + for n in range(4): + files[f"elsewhere/note{n}.md"] = "# Note\n" + # Real drift, not an illustration: the README claims the path in the SAME commit that + # still has it, and a later commit removes it. The fixture's own docstring is emphatic + # that this distinction is the whole rule, and my first attempt at this test ignored it + # and produced a clean run I then asserted was broken. + repo = make_repo( + files, + documented_before={"README.md": "# A project\n\nSee `src/gone.py` for the details.\n"}, + deleted={"src/gone.py": "x = 1\n"}, + ) + main([str(repo)]) + out = capsys.readouterr().out + assert " broken, " in out + assert "4 were never read" in out.strip().splitlines()[-1] + + +def test_full_coverage_adds_no_sentence(make_repo: Callable[..., Path], capsys) -> None: + """A project whose documentation was entirely read must not gain a line telling it so. + The point is to mark a narrow verdict, and a reassurance printed on every clean run is the + kind of noise that teaches a reader to skip the whole block.""" + repo = make_repo({"README.md": README, "src/thing.py": "x = 1\n"}) + main([str(repo)]) + out = capsys.readouterr().out + assert "were never read" not in out