From f022f633db0bb2db5f64139571580cdfb1210004 Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Sun, 20 Sep 2026 14:42:11 -0400 Subject: [PATCH 01/10] research(readability): construct-validity investigation for --readability Reads the code rather than assuming it: --readability is already disclosed as an ordering-only lens (never a grade, not one of --quality-delta's ten gating kinds, folded into --ensemble as a rank not a score). Designs a pairwise + rank-correlation human validation of that ordering claim, and runs the two proxies that need no human label: refactor-commit direction (80 commits, 484 function pairs, lens agrees with the commit's implied direction on only 30.2% of them) and self-consistency under meaning-preserving rewrites (95 mutation attempts across 80 sampled functions, 100% exact tie, as the formula predicts mathematically). Reports both numbers, including the unflattering one, and treats naminglens.h's withdrawn naming-body-mismatch rule as the template for what acting on a failed validation would look like. No src/ changes. Co-Authored-By: Claude Sonnet 5 --- bench/readability_refactor_pairs.py | 279 ++++++++++++++ bench/readability_self_consistency.py | 354 ++++++++++++++++++ .../readability-construct-validity.md | 338 +++++++++++++++++ 3 files changed, 971 insertions(+) create mode 100755 bench/readability_refactor_pairs.py create mode 100755 bench/readability_self_consistency.py create mode 100644 docs/research/readability-construct-validity.md diff --git a/bench/readability_refactor_pairs.py b/bench/readability_refactor_pairs.py new file mode 100755 index 000000000..e162fc2e9 --- /dev/null +++ b/bench/readability_refactor_pairs.py @@ -0,0 +1,279 @@ +#!/usr/bin/env python3 +r"""readability_refactor_pairs.py — proxy (a) of docs/research/readability-construct-validity.md: without +any human label, does `--readability`'s Posnett rank move the direction a refactor/simplify/cleanup commit +message says it should? + +For each commit in this repo's history whose subject matches /refactor|simplify|clean(\s|-)?up/i and that +touches a bounded number of C/C++ source lines, this script extracts each touched .h/.cpp file at the +commit and at its parent, scores every function in each version with the real `--readability` lens (via a +single-file scratch directory, so the lens's own tokenizer and Posnett fit run unmodified), and matches +functions by (file basename, name) that are present in both versions with a DIFFERENT token shape (a +same-shape match means the diff did not touch that specific function's body and is not evidence either +way). It reports what fraction of genuinely-changed functions the lens ranks MORE readable after a commit +whose author called it a refactor — a cheap, label-free stand-in for "does the ranking move the direction a +human editor intended." + +This is a proxy, not ground truth: a commit message is not a readability judgement, and "refactor" commits +sometimes trade readability for something else (performance, a new abstraction boundary) the author valued +more. Read the numbers this way, and see the "what we would like help with" section of the paired doc. + +House rule this script exists under (bench/ANSWERQUALITY.md, bench/BENCHMARK.md): a measurement harness is +a LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say; it is not wired +into test/regression.sh. + +Usage: + bench/readability_refactor_pairs.py # default: this repo, 60 commits, src/ + bench/readability_refactor_pairs.py --max-commits 150 --out /tmp/pairs.tsv + RIPWIRE_BIN=asan/ripwire bench/readability_refactor_pairs.py --json > pairs.json + +Deterministic given a fixed git history and a fixed binary: git log ordering is chronological, the lens +itself is deterministic (docs/ARCHITECTURE.md), and this script does not thread. +""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import re +import shutil +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path +from typing import Optional + +ROOT = Path(__file__).resolve().parent.parent +DEFAULT_BIN = os.environ.get("RIPWIRE_BIN", str(ROOT / "build" / "ripwire")) + +REFACTOR_RE = re.compile(r"refactor|simplify|clean(\s|-)?up", re.IGNORECASE) +SOURCE_EXT = {".h", ".hpp", ".cc", ".cpp", ".cxx"} + +# The Posnett coefficients, copied verbatim from src/readability.h (kPosnettIntercept/Volume/Lines/Entropy) +# so this script can recompute the pre-sigmoid z on FULL precision instead of comparing the CLI's own +# 3-decimal, sigmoid-saturating `posnett=` attribute. z is monotonic in posnett (sigmoid is monotonic), so +# ranking by z ranks identically to ranking by posnett, without the display truncation or the saturation +# that makes several distinct functions all print posnett="0.000" (readability.h's own legend text says +# so). toks= and vocab= are exact integers in the XML, so vol = toks*log2(vocab) is recovered EXACTLY — +# only ent= (2 decimals as printed) stays at display precision, since the per-token frequency table itself +# is not exposed by the verb. +POSNETT_INTERCEPT = 8.87 +POSNETT_VOLUME = -0.033 +POSNETT_LINES = 0.40 +POSNETT_ENTROPY = -1.5 + + +def z_score(vol: float, lines: int, ent: float) -> float: + return POSNETT_INTERCEPT + POSNETT_VOLUME * vol + POSNETT_LINES * lines + POSNETT_ENTROPY * ent + + +@dataclass +class FnPair: + sha: str + subject: str + path: str + name: str + lines_before: int + lines_after: int + toks_before: int + toks_after: int + vol_before: float + vol_after: float + ent_before: float + ent_after: float + posnett_before: float # the CLI's own displayed value — kept for reference, not for comparison + posnett_after: float + + @property + def z_before(self) -> float: + return z_score(self.vol_before, self.lines_before, self.ent_before) + + @property + def z_after(self) -> float: + return z_score(self.vol_after, self.lines_after, self.ent_after) + + @property + def delta(self) -> float: + return self.z_after - self.z_before + + @property + def improved(self) -> bool: + return self.z_after > self.z_before + 1e-9 + + @property + def worsened(self) -> bool: + return self.z_after < self.z_before - 1e-9 + + +def git(*args: str) -> str: + proc = subprocess.run(["git", *args], cwd=str(ROOT), capture_output=True, text=True) + if proc.returncode != 0: + raise RuntimeError(f"git {' '.join(args)} failed: {proc.stderr.strip()}") + return proc.stdout + + +def candidate_commits(max_commits: int, max_changed_lines: int) -> list[tuple[str, str]]: + """(sha, subject) pairs, newest first, whose subject reads as a refactor/simplify/cleanup and whose + total changed-line count over the whole commit is small enough that attributing a per-function score + delta to "the refactor" is defensible rather than noise from an unrelated bulk edit riding along.""" + raw = git( + "log", "--format=%H\x1f%s", "--all", "-i", + "--grep=refactor", "--grep=simplify", "--grep=clean up", "--grep=cleanup", + "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc", + ) + out: list[tuple[str, str]] = [] + for line in raw.splitlines(): + if not line.strip(): + continue + sha, _, subject = line.partition("\x1f") + if not REFACTOR_RE.search(subject): + continue + stat = git("show", "--shortstat", "--format=", sha) + m = re.search(r"(\d+) insertion.*?(\d+) deletion|(\d+) insertion|(\d+) deletion", stat) + changed = 0 + for tok in re.findall(r"(\d+) (?:insertion|deletion)s?\(\+?-?\)?", stat): + changed += int(tok) + if changed == 0 or changed > max_changed_lines: + continue + out.append((sha, subject)) + if len(out) >= max_commits: + break + return out + + +def touched_source_files(sha: str) -> list[str]: + raw = git("show", "--name-only", "--format=", sha) + return [p for p in raw.splitlines() if p and Path(p).suffix in SOURCE_EXT and p.startswith("src/")] + + +def show_file(rev: str, path: str) -> Optional[str]: + proc = subprocess.run(["git", "show", f"{rev}:{path}"], cwd=str(ROOT), capture_output=True, text=True) + if proc.returncode != 0: + return None + return proc.stdout + + +def score_file(binary: str, scratch_dir: Path, basename: str, text: str) -> dict[str, tuple]: + """Write `text` as the sole file in an otherwise-empty directory, run --readability over it, and + return {function_name: (lines, toks, vol_exact, ent, posnett)}. A single-file directory is enough — + the lens is per-function and needs no cross-file resolution (readability.h's own header note).""" + if scratch_dir.exists(): + shutil.rmtree(scratch_dir) + scratch_dir.mkdir(parents=True) + (scratch_dir / basename).write_text(text, encoding="utf-8", errors="surrogateescape") + proc = subprocess.run( + [binary, str(scratch_dir), "--readability", "--limit=100000"], + capture_output=True, text=True, timeout=60, + ) + if proc.returncode != 0: + raise RuntimeError(f"--readability exited {proc.returncode} on {scratch_dir}: {proc.stderr.strip()}") + root = ET.fromstring(proc.stdout) + rows: dict[str, tuple] = {} + for fn in root.findall("fn"): + name = fn.get("n") + # Same-named overload/specialization collisions are rare in one file; keep the first, since a + # scratch single-file rescoring only needs A representative match, not perfect overload resolution. + if name not in rows: + toks = int(fn.get("toks")) + vocab = int(fn.get("vocab")) + vol_exact = toks * math.log2(vocab) if vocab > 0 else 0.0 # readability.h's own formula, integer inputs + rows[name] = (int(fn.get("lines")), toks, vol_exact, float(fn.get("ent")), float(fn.get("posnett"))) + return rows + + +def collect_pairs(binary: str, max_commits: int, max_changed_lines: int, scratch: Path) -> list[FnPair]: + pairs: list[FnPair] = [] + commits = candidate_commits(max_commits, max_changed_lines) + print(f"# {len(commits)} candidate refactor/simplify/cleanup commits (max_changed_lines={max_changed_lines})", file=sys.stderr) + before_dir = scratch / "before" + after_dir = scratch / "after" + for sha, subject in commits: + for path in touched_source_files(sha): + before_text = show_file(f"{sha}^", path) + after_text = show_file(sha, path) + if before_text is None or after_text is None: + continue # file added or deleted in this commit — no pair to compare + basename = Path(path).name + try: + before_rows = score_file(binary, before_dir, basename, before_text) + after_rows = score_file(binary, after_dir, basename, after_text) + except RuntimeError as exc: + print(f"# skip {sha[:10]} {path}: {exc}", file=sys.stderr) + continue + for name, (lb, tb, volb, entb, pb) in before_rows.items(): + if name not in after_rows: + continue + la, ta, vola, enta, pa = after_rows[name] + if (lb, tb) == (la, ta): + continue # identical shape — this function's body was not touched by the diff + pairs.append(FnPair(sha, subject, path, name, lb, la, tb, ta, volb, vola, entb, enta, pb, pa)) + return pairs + + +def summarize(pairs: list[FnPair]) -> dict: + improved = sum(1 for p in pairs if p.improved) + worsened = sum(1 for p in pairs if p.worsened) + tied = len(pairs) - improved - worsened + n_directional = improved + worsened + frac_improved = improved / n_directional if n_directional else float("nan") + mean_delta = sum(p.delta for p in pairs) / len(pairs) if pairs else float("nan") + return { + "pairs": len(pairs), + "improved": improved, + "worsened": worsened, + "tied_within_1e-9": tied, + "frac_improved_of_directional": frac_improved, + "mean_z_delta": mean_delta, + } + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=DEFAULT_BIN) + ap.add_argument("--max-commits", type=int, default=60) + ap.add_argument("--max-changed-lines", type=int, default=400, + help="skip commits whose total insertion+deletion count exceeds this (default 400) — " + "keeps the compared functions attributable to the named refactor") + ap.add_argument("--out", default=None, help="write the per-pair TSV here (default: stdout table only)") + ap.add_argument("--json", action="store_true", help="print the summary as JSON instead of a table") + ap.add_argument("--scratch", default=None, help="scratch directory (default: a fresh temp dir)") + args = ap.parse_args() + + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin} — build it first (see CLAUDE.md)", file=sys.stderr) + return 2 + + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_readpairs_")) + try: + pairs = collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) + + summary = summarize(pairs) + + if args.out: + with open(args.out, "w", encoding="utf-8") as f: + f.write("sha\tsubject\tpath\tname\tz_before\tz_after\tz_delta\tposnett_before\tposnett_after\tlines_before\tlines_after\ttoks_before\ttoks_after\n") + for p in pairs: + f.write(f"{p.sha}\t{p.subject}\t{p.path}\t{p.name}\t{p.z_before:.6f}\t{p.z_after:.6f}\t{p.delta:.6f}\t{p.posnett_before:.3f}\t{p.posnett_after:.3f}\t{p.lines_before}\t{p.lines_after}\t{p.toks_before}\t{p.toks_after}\n") + print(f"# wrote {len(pairs)} pairs to {args.out}", file=sys.stderr) + + if args.json: + print(json.dumps(summary, indent=2)) + else: + print(f"pairs (changed-function before/after) : {summary['pairs']}") + print(f"lens ranks AFTER more readable (z increased) : {summary['improved']}") + print(f"lens ranks AFTER less readable (z decreased) : {summary['worsened']}") + print(f"exact tie (z unchanged, to 1e-9) : {summary['tied_within_1e-9']}") + frac = summary["frac_improved_of_directional"] + print(f"fraction improved, of the directional pairs : {frac:.3f}" if frac == frac else "fraction improved: n/a (no directional pairs)") + print(f"mean z delta (after - before; +delta=more readable) : {summary['mean_z_delta']:.4f}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/bench/readability_self_consistency.py b/bench/readability_self_consistency.py new file mode 100755 index 000000000..1747d3aba --- /dev/null +++ b/bench/readability_self_consistency.py @@ -0,0 +1,354 @@ +#!/usr/bin/env python3 +r"""readability_self_consistency.py — proxy (b) of docs/research/readability-construct-validity.md: without +any human label, does `--readability` agree with ITSELF across two mechanical rewrites that should not +change how readable a function is — renaming one local to a fresh, non-colliding name, and reordering two +adjacent, mutually-independent simple statements? + +Method: sample functions from this repo's own `src` at percentile positions across the lens's own ranked +list (least-readable-first), so the sample spans the whole distribution rather than only the worst decile +--readability itself prints by default. For each sampled function, re-score its OWN whole file unmodified +in a single-file scratch directory (so the comparison baseline takes the identical measurement path the +mutant does — no cross-file ingest differences to explain away), apply one mutation, re-score the mutated +whole file, and compare. + +WHAT THIS TEST ACTUALLY MEASURES, STATED PLAINLY. Halstead volume and Shannon entropy are functions of the +MULTISET of token (class, frequency) pairs, not of token identity or statement order — readability.h's own +determinism note says the entropy sum runs over a token-text-sorted vector precisely because order must not +reach the output. So for a rename to a fresh (non-colliding) identifier, or a reorder of two independent +statements, ZERO change in `vol=`/`ent=`/`posnett=` is what the formula predicts mathematically, not a +finding this script discovers. A pass here mostly certifies IMPLEMENTATION correctness (no stray +non-determinism, no order leak) rather than any claim about human-perceived readability invariance — and it +is honest about that limit rather than overselling a trivial pass as construct validity. The genuinely +informative reading is the FAILURE case: any pair that does NOT tie exactly is either an extraction/mutation +bug in this harness or a real order/identity sensitivity in the lens worth filing. The other honest reading +is what this test structurally CANNOT catch: because the formula is blind to identifier length and to +statement order by construction, it also cannot reward `numberOfActiveConnections` over `n`, or a +better-ordered proof over a worse one — see the "what we would like help with" section of the paired doc. + +House rule this script exists under (bench/ANSWERQUALITY.md, bench/BENCHMARK.md): a measurement harness is +a LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say. + +Usage: + bench/readability_self_consistency.py # default: this repo's src/, 30 samples + bench/readability_self_consistency.py --samples 60 --seed 7 + RIPWIRE_BIN=asan/ripwire bench/readability_self_consistency.py --json > consistency.json +""" + +from __future__ import annotations + +import argparse +import json +import os +import random +import re +import shutil +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path +from typing import Optional + +ROOT = Path(__file__).resolve().parent.parent +DEFAULT_BIN = os.environ.get("RIPWIRE_BIN", str(ROOT / "build" / "ripwire")) +DEFAULT_CORPUS = "src" + +# A fixed, deliberately weird suffix: astronomically unlikely to already occur as a real identifier in this +# tree, so a rename to `_mutrn7k` cannot collide with an existing token and silently change vocab=. +RENAME_SUFFIX = "_mutrn7k" + +CPP_KEYWORDS = { + "if", "else", "for", "while", "do", "switch", "case", "default", "break", "continue", "return", + "const", "static", "inline", "constexpr", "auto", "void", "int", "bool", "char", "double", "float", + "struct", "class", "namespace", "using", "typename", "template", "public", "private", "protected", + "true", "false", "nullptr", "new", "delete", "this", "sizeof", "noexcept", "override", "virtual", + "std", "size_t", "uint32_t", "uint64_t", "int32_t", "int64_t", "std::size_t", +} + +IDENT_RE = re.compile(r"\b[a-z_][a-zA-Z0-9_]*\b") +IDENT_ANY_RE = re.compile(r"[A-Za-z_]\w*") +# A conservative "simple statement" shape: one line, ending `;`, no braces (so it is not secretly a whole +# block), with exactly one PLAIN `=` — not `==`/`!=`/`<=`/`>=` and not a compound `+=`/`-=`/… . Matches both +# a bare reassignment (`x = y + z;`) and a one-line typed declaration (`const Foo& x = expr;`), which this +# house style uses constantly (CONTRIBUTING.md Allman style, one statement per line). +SIMPLE_STMT_RE = re.compile(r"^(?P\s*)(?P[^{};]+);\s*$") +COMPOUND_EQ_PREV = set("=!<>+-*/%&|^") + + +def parse_simple_stmt(line: str) -> Optional[tuple[str, str, str]]: + """(indent, lhs_variable_name, rhs_text) for a line this script is willing to reorder, or None. The LHS + name is the LAST identifier immediately before the plain `=` — for a typed declaration that is the + variable, not the type.""" + m = SIMPLE_STMT_RE.match(line) + if not m: + return None + body = m.group("body") + plain_eqs = [i for i, ch in enumerate(body) + if ch == "=" and (i == 0 or body[i - 1] not in COMPOUND_EQ_PREV) and body[i + 1:i + 2] != "="] + if len(plain_eqs) != 1: + return None + i = plain_eqs[0] + lhs_part, rhs_part = body[:i].rstrip(), body[i + 1:].strip() + if "(" in lhs_part or "[" in lhs_part or "->" in lhs_part: + return None # a call, subscript or member target — too easy to get independence wrong + idents = IDENT_ANY_RE.findall(lhs_part) + if not idents or idents[-1] in CPP_KEYWORDS: + return None + return m.group("indent"), idents[-1], rhs_part + + +@dataclass +class FnRow: + path: str + line: int + name: str + lines: int + posnett: float + rank_index: int # position in the lens's own least-readable-first ordering + + +def run_readability(binary: str, target: str, cwd: Optional[Path] = None) -> list[FnRow]: + proc = subprocess.run( + [binary, target, "--readability", "--limit=100000"], + capture_output=True, text=True, timeout=120, cwd=str(cwd) if cwd else None, + ) + if proc.returncode != 0: + raise RuntimeError(f"--readability exited {proc.returncode} on {target}: {proc.stderr.strip()}") + root = ET.fromstring(proc.stdout) + rows = [] + for i, fn in enumerate(root.findall("fn")): + p, _, ln = fn.get("p").rpartition(":") + rows.append(FnRow(p, int(ln), fn.get("n"), int(fn.get("lines")), float(fn.get("posnett")), i)) + return rows + + +def score_single_file(binary: str, scratch_dir: Path, basename: str, text: str) -> dict[str, dict]: + if scratch_dir.exists(): + shutil.rmtree(scratch_dir) + scratch_dir.mkdir(parents=True) + (scratch_dir / basename).write_text(text, encoding="utf-8", errors="surrogateescape") + proc = subprocess.run( + [binary, str(scratch_dir), "--readability", "--limit=100000"], + capture_output=True, text=True, timeout=60, + ) + if proc.returncode != 0: + raise RuntimeError(f"--readability exited {proc.returncode} on {scratch_dir}: {proc.stderr.strip()}") + root = ET.fromstring(proc.stdout) + out: dict[str, dict] = {} + for fn in root.findall("fn"): + name = fn.get("n") + if name not in out: + out[name] = { + "lines": int(fn.get("lines")), "toks": int(fn.get("toks")), "vocab": int(fn.get("vocab")), + "vol": float(fn.get("vol")), "ent": float(fn.get("ent")), "posnett": float(fn.get("posnett")), + } + return out + + +def stratified_sample(rows: list[FnRow], n: int, rng: random.Random) -> list[FnRow]: + """n rows spread across the WHOLE ranked distribution (percentile bins with one jittered pick each), not + just the worst decile the verb shows by default — a consistency check only worth trusting if it covers + functions the lens currently calls readable as well as ones it calls unreadable.""" + if len(rows) <= n: + return list(rows) + picked = [] + bin_size = len(rows) / n + for i in range(n): + lo = int(i * bin_size) + hi = max(lo + 1, int((i + 1) * bin_size)) + hi = min(hi, len(rows)) + picked.append(rows[rng.randrange(lo, hi)]) + return picked + + +def extract_span(file_lines: list[str], start_line: int, span: int) -> Optional[tuple[int, int]]: + """1-indexed inclusive [start_line, start_line+span-1], clamped; None if out of range.""" + lo = start_line - 1 + hi = lo + span + if lo < 0 or hi > len(file_lines) or lo >= hi: + return None + return lo, hi + + +def try_rename(span_lines: list[str], fn_name: str) -> Optional[tuple[list[str], str]]: + """Pick a lowercase identifier appearing >=2 times in the span, not a keyword, not the function's own + name, not already ending in RENAME_SUFFIX, whole-word-replace every occurrence. Returns (new_lines, + renamed_identifier) or None if no eligible candidate exists.""" + text = "\n".join(span_lines) + counts: dict[str, int] = {} + for m in IDENT_RE.finditer(text): + tok = m.group(0) + if tok in CPP_KEYWORDS or tok == fn_name or len(tok) < 2 or tok.endswith(RENAME_SUFFIX): + continue + counts[tok] = counts.get(tok, 0) + 1 + candidates = [t for t, c in counts.items() if c >= 2] + if not candidates: + return None + # Deterministic pick: the most frequent candidate, ties broken lexically — reproducible across runs. + candidates.sort(key=lambda t: (-counts[t], t)) + victim = candidates[0] + new_name = victim + RENAME_SUFFIX + if new_name in text: + return None # pathological collision; skip rather than risk a silent vocab change + pattern = re.compile(r"\b" + re.escape(victim) + r"\b") + new_text = pattern.sub(new_name, text) + return new_text.split("\n"), victim + + +def try_reorder(span_lines: list[str]) -> Optional[tuple[list[str], int]]: + """Find the first adjacent pair of simple statement lines (parse_simple_stmt), same indentation, + different LHS name, where neither RHS mentions the other's LHS as a whole word (the independence + heuristic) — and swap the two full lines.""" + for i in range(len(span_lines) - 1): + p1 = parse_simple_stmt(span_lines[i]) + p2 = parse_simple_stmt(span_lines[i + 1]) + if not p1 or not p2: + continue + indent1, lhs1, rhs1 = p1 + indent2, lhs2, rhs2 = p2 + if indent1 != indent2 or lhs1 == lhs2: + continue + if re.search(r"\b" + re.escape(lhs1) + r"\b", rhs2) or re.search(r"\b" + re.escape(lhs2) + r"\b", rhs1): + continue + out = list(span_lines) + out[i], out[i + 1] = out[i + 1], out[i] + return out, i + return None + + +@dataclass +class MutationResult: + path: str + name: str + kind: str # "rename" | "reorder" + detail: str + baseline: dict + mutant: dict + + def ties(self) -> bool: + return (abs(self.baseline["vol"] - self.mutant["vol"]) < 1e-6 + and abs(self.baseline["ent"] - self.mutant["ent"]) < 1e-6 + and self.baseline["posnett"] == self.mutant["posnett"]) + + +def run(binary: str, corpus: str, n_samples: int, seed: int, scratch: Path) -> list[MutationResult]: + rng = random.Random(seed) + all_rows = run_readability(binary, corpus, cwd=ROOT) + print(f"# {len(all_rows)} functions measured in {corpus}; sampling {n_samples} across the ranked distribution", file=sys.stderr) + sample = stratified_sample(all_rows, n_samples, rng) + + results: list[MutationResult] = [] + baseline_dir = scratch / "baseline" + mutant_dir = scratch / "mutant" + for row in sample: + # `p=` in the lens's own output is relative to the scanned root (its legend says so verbatim); + # rejoin against the corpus argument this run used to get a real path. + abspath = ROOT / corpus / row.path + if not abspath.is_file(): + continue + whole = abspath.read_text(encoding="utf-8", errors="surrogateescape") + file_lines = whole.split("\n") + span = extract_span(file_lines, row.line, row.lines) + if span is None: + continue + lo, hi = span + span_lines = file_lines[lo:hi] + span_text = "\n".join(span_lines) + if row.name not in span_text: + print(f"# skip {row.path}:{row.line} {row.name} — extracted span does not mention its own name (line-mapping drift)", file=sys.stderr) + continue + basename = Path(row.path).name + + try: + baseline_scores = score_single_file(binary, baseline_dir, basename, whole) + except RuntimeError as exc: + print(f"# skip {row.path} baseline: {exc}", file=sys.stderr) + continue + if row.name not in baseline_scores: + continue + base_row = baseline_scores[row.name] + + renamed = try_rename(span_lines, row.name) + if renamed is not None: + new_span, victim = renamed + new_lines = file_lines[:lo] + new_span + file_lines[hi:] + mutant_text = "\n".join(new_lines) + try: + mutant_scores = score_single_file(binary, mutant_dir, basename, mutant_text) + if row.name in mutant_scores: + results.append(MutationResult(row.path, row.name, "rename", f"{victim} -> {victim}{RENAME_SUFFIX}", + base_row, mutant_scores[row.name])) + except RuntimeError as exc: + print(f"# skip {row.path} rename: {exc}", file=sys.stderr) + + reordered = try_reorder(span_lines) + if reordered is not None: + new_span, at = reordered + new_lines = file_lines[:lo] + new_span + file_lines[hi:] + mutant_text = "\n".join(new_lines) + try: + mutant_scores = score_single_file(binary, mutant_dir, basename, mutant_text) + if row.name in mutant_scores: + results.append(MutationResult(row.path, row.name, "reorder", f"swap span-local lines {at},{at+1}", + base_row, mutant_scores[row.name])) + except RuntimeError as exc: + print(f"# skip {row.path} reorder: {exc}", file=sys.stderr) + + return results + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=DEFAULT_BIN) + ap.add_argument("--corpus", default=DEFAULT_CORPUS) + ap.add_argument("--samples", type=int, default=30) + ap.add_argument("--seed", type=int, default=20260918) + ap.add_argument("--out", default=None) + ap.add_argument("--json", action="store_true") + ap.add_argument("--scratch", default=None) + args = ap.parse_args() + + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin} — build it first (see CLAUDE.md)", file=sys.stderr) + return 2 + + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_readconsist_")) + try: + results = run(args.bin, args.corpus, args.samples, args.seed, scratch) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) + + by_kind: dict[str, list[MutationResult]] = {"rename": [], "reorder": []} + for r in results: + by_kind[r.kind].append(r) + + summary = {} + for kind, rs in by_kind.items(): + ties = sum(1 for r in rs if r.ties()) + summary[kind] = {"attempted": len(rs), "exact_tie": ties, "diverged": len(rs) - ties} + + if args.out: + with open(args.out, "w", encoding="utf-8") as f: + f.write("kind\tpath\tname\tdetail\ttie\tvol_before\tvol_after\tent_before\tent_after\tposnett_before\tposnett_after\n") + for r in results: + f.write(f"{r.kind}\t{r.path}\t{r.name}\t{r.detail}\t{int(r.ties())}\t{r.baseline['vol']:.4f}\t{r.mutant['vol']:.4f}\t{r.baseline['ent']:.4f}\t{r.mutant['ent']:.4f}\t{r.baseline['posnett']:.3f}\t{r.mutant['posnett']:.3f}\n") + print(f"# wrote {len(results)} mutation results to {args.out}", file=sys.stderr) + + if args.json: + print(json.dumps(summary, indent=2)) + else: + for kind, s in summary.items(): + print(f"{kind:8s} attempted={s['attempted']:3d} exact_tie={s['exact_tie']:3d} diverged={s['diverged']:3d}") + for r in results: + if not r.ties(): + print(f" DIVERGED [{r.kind}] {r.path} {r.name} ({r.detail}): " + f"vol {r.baseline['vol']:.2f}->{r.mutant['vol']:.2f} " + f"ent {r.baseline['ent']:.2f}->{r.mutant['ent']:.2f} " + f"posnett {r.baseline['posnett']:.3f}->{r.mutant['posnett']:.3f}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md new file mode 100644 index 000000000..bec5163c1 --- /dev/null +++ b/docs/research/readability-construct-validity.md @@ -0,0 +1,338 @@ +# Readability construct validity: what the lens claims, and how we would check it + +Status: **investigation, not a conclusion.** This document states what `--readability` actually computes +and actually claims (read from the code, not assumed), designs a validation of its *ordering* against human +judgment that we have not yet run, and runs the two validations that need no human labels at all — reporting +their numbers whether or not they are flattering. It ends with what we would like help with from readability +researchers who study exactly this gap. + +Two published results shaped where ripwire's readability lens sits, and both are **design influences, not +implementations** — nothing in `src/readability.h` runs a model or a judge: + +- LLM judges of code readability lean on surface features rather than the construct itself — part of why + ripwire's readability lens is a deterministic, closed-form formula (Halstead volume, token entropy, the + Posnett sigmoid fit) instead of a model call, and why the underlying evidence is disclosed (`vol=`, `ent=`, + `posnett=`) rather than a single opaque number. +- A study of prompt-side style constraints on LLM code generation found they help but plateau — part of why + ripwire's readability-adjacent gate (`--quality-delta`'s `verbosity` kind) runs **after** each edit, against + the actual diff, rather than living inside a generation prompt as one more instruction competing with the + rest of the prompt for effect. + +Both are cited in this repo's own research log, `DESIGN_READABILITY_METRICS.md` §7, as the "Readability +Spectrum" prompt-constraint finding and as "LLM self-judges … fixate on surface features (CoReEval)" — +see the reference list at the end of this document for both arXiv identifiers. That design log already states +the honesty bound this document expands on: *"Everything here is a lens, never a verdict."* This document is +the follow-through on that bound — checking it empirically rather than repeating it. + +## 1. What we actually compute, and what we actually claim (from the code) + +`--readability` (`src/readability.h`) is a single closed-form pass per function or method: + +- **V — Halstead volume**: `V = N · log2(η)`, `N` = operator+operand token count, `η` = distinct tokens. +- **E — token entropy**: Shannon entropy (bits) of the definition's own token-frequency distribution. +- **L — lines**: the physical line span of the definition (`Symbol::loc`), signature included. +- **P — Posnett score**: `P = sigmoid(8.87 − 0.033·V + 0.40·L − 1.5·E)` — the coefficients published in + Posnett, Hindle & Devanbu, *A Simpler Model of Software Readability*, MSR 2011. + +Three facts about how this is used, verified against the source rather than assumed: + +**It is emitted, and used elsewhere in the tool, as an ORDERING, never as a grade.** The verb sorts rows +*least readable first* and says so in its own legend (`src/readability.h`, `kReadabilityLegend`): + +> "P was fitted on snippets of 20 lines or fewer: read the ORDER, not the number, and never as a grade." + +The header comment states the reason in more detail, and names the literature that forced this framing +rather than a house preference: *"Scalabrino ASE'17 and Trockman MSR'18 both find no readability metric +correlates strongly with measured understandability; Fakhoury ICPC'19 finds the classic models miss real +readability-improving commits. So P is never a grade, never a gate, and never a verdict — it orders a +worklist."* The published `README.md` repeats the same claim as an external-facing promise, in a table of +lenses deliberately kept *outside* the tool's evidence-weighted panel join. Its row for `--readability` +reads, across that row's "why beside the join" and "on this repo" cells: *"the fitted score saturates past 20 +lines — only the ordering is meaningful, and an ordering cannot vote in a count"* / *"ordering only, never a +grade."* + +**It is not one of `--quality-delta`'s ten gating kinds.** The per-edit gate's kinds are `complexity`, +`verbosity`, `nesting`, `params`, `duplication`, `dead-code`, `api-surface`, `error-masking`, +`short-horizon-churn` and `new-clone-of-reused-helper` (`src/quality.h:4565` and the kind dispatch around +it). None of them is Halstead volume, token entropy or the Posnett score. The one readability-*adjacent* +kind is `verbosity`, and reading its implementation shows exactly what it measures and what it does not: +`verbosity` is **CODE line count only** (`src/quality.h`, `locBySym`, `codeLocByNode` — "Q-DIAL-3: blank and +comment-only lines are not debt"). It shares one input (`L`) with the Posnett formula and none of the other +two (`V`, `E`). So the per-edit gate that actually blocks a merge never sees the entropy or Halstead-volume +half of the readability lens at all — it sees a plain, un-weighted line count. + +**The only place the Posnett rank feeds a joined judgment is `--ensemble`, and even there it is kept +ordinal, not additive.** `src/ensemble.h` folds `--readability`'s rank into its "structural" evidence family +alongside complexity/LOC/nesting/params, and its own header comment states why the five are one family and +not five independent votes: *"ccx, loc, nest, params and the Posnett score all track SIZE — Posnett's own fit +is literally linear in L. Counting them as five agreeing witnesses is the Maintainability-Index failure +(§3.10): re-weighting one signal and calling it five."* — and separately, on why the Posnett rank and churn +are handled differently from the other four: *"The other two signals (Posnett readability, churn) are +RANKINGS whose own authors publish no defensible absolute cut: --readability's header says in so many words +to read the ORDER, not the number. The only honest predicate on an ordinal signal is an ordinal cut, so each +fires for the WORST DECILE of its own ranking."* `--ensemble` records only a symbol's position in that +worst-decile prefix (`rrank=`), never `posnett=` itself — there is, by the same comment's own words a few +lines later, "no composite number anywhere in this verb, by contract." + +**So: does the tool present readability as an absolute grade anywhere?** No surface we found does. The one +place a reader could plausibly *read* it that way — a raw `posnett=` value on an individual row of +`--readability`'s own output — carries the legend's own correction directly above it every time it is +printed ("read the ORDER, not the number, and never as a grade"), and the value visibly saturates to +`0.000` for a large share of real functions (see §3), which is itself a standing, unavoidable reminder that +the number is not meant to be read on its own. We consider the code's claim to already be the *honest* one: +ordering-only, disclosed as ordering-only, at every point of contact. The rest of this document is about +whether that ordering claim itself survives contact with a human judgment — which is a strictly harder bar +than "is the code's rhetoric honest," and one the code freely admits it has not cleared (Scalabrino/Trockman/ +Fakhoury are cited as open problems, not as solved ones). + +## 2. A validation of the ORDERING, not the score + +Reproducibility (the same input always yields the same rank) is not construct validity (the rank means what +we say it means). The formula could be perfectly deterministic and still rank functions in an order no human +reader would recognize. Here is the validation we think would settle that, specified concretely enough to run: + +**Unit of comparison: pairs, not scores.** Ask a rater "which of these two is more readable," never "rate +this snippet 1–5" — the literature's own ground-truth datasets are pairwise-derived or scale-based and Vitale +(2025, cited in `DESIGN_READABILITY_METRICS.md` §7) found up to a third of classic scale labels +self-contradictory on reread. A forced pairwise choice is cheaper to collect, cheaper to get consistent, and +is exactly the shape `--readability`'s own claim needs checking against: it emits an order, so validate the +order. + +**Where pairs come from (three tiers, cheapest first):** + +1. **This repo's own git history** — a commit whose message says refactor/simplify/cleanup gives a free + (before, after) pair of the same function, no synthetic construction needed. This is what §3's proxy (a) + already mines, at zero marginal data-collection cost. +2. **`bench/external/arb` and `bench/external/swex`** — this repo already vendors multiple external-repo + snapshots and commit histories for other evaluation harnesses (agentic-benchmark corpora, not readability + ones). The same commit-message mining in §3's script generalizes to any of them by pointing `--root` at a + different checkout — more languages, more authors, no ripwire-specific style bias. We did not run this for + the current round (time-boxed to this repo, see §3) but the script takes a `--bin`/corpus-root pair for + exactly this reason. +3. **Synthetic pairs from `--readability`'s own ranking** — take two functions already far apart in the + lens's own order (one from the worst decile, one from the best) and ask a rater to confirm or reject the + implied direction. Cheapest to collect, weakest signal (it can only confirm the lens agrees with itself at + the extremes, which §3 proxy (b) already establishes for free without a human). + +**How a human judgment would be collected.** Blinded, randomized left/right presentation (no file path, no +`posnett=`, no commit message); each pair rated independently by at least three raters, majority vote as the +pair's label, inter-rater agreement reported alongside the correlation (a low agreement number is itself a +finding, per Vitale). Raters should not be told which side the lens preferred — Piantadosi et al.'s "readable +state flip" framing (cited in `DESIGN_READABILITY_METRICS.md` §6) is a reasonable model for how to phrase the +question without anchoring the rater on a metric. + +**The statistic.** Percentage pairwise agreement between the lens's implied direction and the majority human +label, plus a rank correlation (Kendall's τ or Spearman's ρ) over any batch of pairs drawn from one shared +ranked list, since τ is exactly a transform of pairwise agreement and gives readers a standard number to +compare against the published literature's own correlations for competing metrics. + +**What would count as success, and what would make us withdraw or demote the lens.** We propose three bands, +calibrated against the literature already cited in this repo (Scalabrino ASE'17 and Trockman MSR'18 found +*no* classic metric correlates strongly with measured understandability — so we are not calibrating against +"strong correlation is achievable," we are calibrating against "is this metric earning the narrow claim it +actually makes"): + +- **τ (or equivalent pairwise agreement) comfortably above chance and stable across a second, disjoint + sample** → keep exactly as-is: an ordering signal, disclosed as such, feeding `--ensemble`'s rank-only + join and nothing stronger. +- **Weak but directionally consistent, or consistent on some function shapes and not others (e.g., holds for + size-dominated differences, fails when two functions are close in size but differ in naming/structure)** → + narrow the claim further and say so in the legend — e.g. disclose that the lens is known to track length + more than it tracks anything len-independent, which §3's proxy (b) already shows structurally (see the + "what this cannot catch" paragraph there). +- **No better than chance, or a sign flip on a held-out language/corpus** → withdraw the ranking claim from + any joined surface (`--ensemble`'s `rrank=`) and keep `--readability` only as a standalone, clearly-labeled + "how does this formula see the codebase" report — the same demotion path `naminglens.h` already used once + (§4). + +We have not run the human-rated arm. It needs raters we do not have in this pass; §5 names it as the thing we +would like the most external help with. + +## 3. What we ran WITHOUT human labels + +Two proxies need no rater at all. Both are implemented as re-runnable scripts against the real +`./build/ripwire --readability` binary — not a reimplementation of the formula — so what they measure is +the shipped lens, not a paper description of it. + +### 3a. Refactor-commit direction (`bench/readability_refactor_pairs.py`) + +For every commit in this repo's history whose subject reads as refactor/simplify/cleanup (case-insensitive, +`refactor|simplify|clean(\s|-)?up`) and whose total changed-line count is ≤400 (so a per-function delta stays +attributable to the named refactor rather than to an unrelated bulk edit riding along in the same commit), +the script extracts every touched `.h`/`.cpp` file at the commit and at its parent, scores each version with +the real lens in a single-file scratch directory, and keeps the functions present in both versions with a +**different** token shape (same shape ⇒ this diff did not touch that function's body ⇒ no evidence either +way). Comparison uses the pre-sigmoid `z` score recomputed at full precision from the CLI's own +integer `toks=`/`vocab=` and its `ent=`/`lines=` — not the CLI's own 3-decimal `posnett=` attribute, which +saturates to a repeated `"0.000"` for most real functions (see the tie count below) and would make most pairs +falsely indistinguishable. `z` is monotonic in `posnett` (same sigmoid), so ranking by `z` ranks identically +to ranking by `posnett` without the display truncation. + +**Run** (`--max-commits 80 --max-changed-lines 400`, the numbers below): + +| | | +|---|---| +| candidate refactor/simplify/cleanup commits | 80 (78 contributed ≥1 usable pair) | +| function pairs (changed shape, matched by name) | 484 | +| lens ranks AFTER more readable (`z` increased) | 146 (30.2% of the 484 directional pairs) | +| lens ranks AFTER less readable (`z` decreased) | 338 (69.8%) | +| exact tie | 0 | +| mean Δz (after − before, + = more readable) | **+1.89** | +| median Δz | **−0.73** | +| pairs with \|Δz\| > 20 (large swings) | 27 / 484 (5.6%) | + +**Read this plainly, including that it is not flattering.** On a majority (70%) of the function pairs a human +called a refactor, the lens's own ranking moved the *wrong* direction — it scored the after-version as *less* +readable. The mean is positive only because a small number of large swings (5.6% of pairs, all from commits +that genuinely shrank a function by splitting work into named helpers) pull it there; the median, which a +skewed distribution like this one should be read against, is negative. This lines up with exactly the +literature the lens's own code already cites as a reason for caution — Fakhoury ICPC'19's finding that +classic readability models "miss real-world readability-improvement commits" is not a hypothetical risk here, +it is what this run measured on ripwire's own history. + +**Two worked examples, to show what is and is not driving the number.** The eight largest positive swings +are all commits whose stated purpose was extraction — moving a block of logic out into named helper +functions, which mechanically shortens the function the lens is scoring: `buildFieldNarrowTables` +(114→50 lines), `computeSnapshot` (116→61), `buildScopedRecvDecls` (61→15), `hasNetExfilShape` (59→17), +`buildExternalVetoTables` (147→86), `resolve` in `src/elixir_resolve.h` (57→17), `parseAsan` (99→21) and +`runSkipped` (71→45) — a 25%–75% line-count cut in every case. The lens's length-sensitivity is doing +exactly the intuitive thing there. The single +largest *negative* swing is the opposite shape of edit: commit `15af398e` +(`refactor(L1-fix): keep existing contracts; readings spell no element markup`) inlined a one-line forwarding +wrapper (`classify()`, 5 lines) into what had been a same-named sibling implementation, producing one 126-line +function where there had been a 5-line indirection plus a separate body. The commit message is about +preserving an API contract, not about readability, and a human reader might reasonably call the *pre*-commit +two-function shape less readable (an unexplained one-line forwarder) than the *post*-commit single function — +the opposite of what the lens's length term rewards. We are not correcting for this by hand; it is exactly +the kind of case a pairwise human study (§2) would need to arbitrate, and we would be misrepresenting the +proxy if we quietly excluded it. + +Re-run: `bench/readability_refactor_pairs.py --max-commits 80 --out pairs.tsv` (defaults to this repo, +`build/ripwire`; deterministic given a fixed git history and a fixed binary). + +### 3b. Self-consistency under meaning-preserving rewrites (`bench/readability_self_consistency.py`) + +Two mechanical rewrites that should not change how readable a function is: renaming one local identifier to a +fresh name that collides with nothing else in the function, and swapping two adjacent, mutually-independent +simple statements (`lhs = rhs;`, including one-line typed declarations, in this repo's one-statement-per-line +house style). 80 functions were sampled at even percentile positions across `--readability`'s own +least-readable-first ranking of this repo's `src/` (4,373 functions measured), so the sample spans the whole +distribution rather than only the worst decile the verb shows by default. Each sampled function's own whole +file is re-scored unmutated as the baseline (same single-file measurement path the mutant takes, so there is +no cross-file ingest difference to explain away), then rescored after one mutation. + +| mutation | attempted | exact tie (`vol=`, `ent=`, `posnett=` all unchanged) | diverged | +|---|---|---|---| +| rename (fresh, non-colliding identifier) | 75 / 80 | 75 | 0 | +| reorder (two adjacent independent statements) | 20 / 80 | 20 | 0 | + +**State plainly what this does and does not show.** Halstead volume and Shannon entropy are functions of the +*multiset* of (token, frequency) pairs — not of token identity or statement order (`readability.h`'s own +determinism note: the entropy sum runs over a token-text-**sorted** vector specifically so iteration order +cannot reach the output). So exact invariance under a non-colliding rename or an independent-statement +reorder is what the formula predicts *mathematically*, not a discovery — a 100% tie rate here mostly certifies +that the shipped implementation has no stray non-determinism or order leak, which is a real and worth-having +guarantee, but it is an implementation-correctness result, not a readability-construct-validity result. The +informative failure mode this proxy could have caught — and did not — is any pair that does NOT tie exactly, +which would point at either a bug in this harness's mutation or a genuine order/identity sensitivity in the +lens worth filing as a defect. + +**The same invariance is also a disclosed limitation, not only a safety net.** Because the formula is +mathematically blind to identifier length and to statement order, it is *structurally incapable* of +rewarding `numberOfActiveConnections` over `n`, or a well-ordered proof sketch over a scrambled one — both +of which most human readers would call a real readability difference. That is a construct-validity gap this +proxy makes visible without needing a single human label: whatever the lens is measuring, it provably is not +measuring those two things, by the formula's own definition. + +**Coverage caveat, reported rather than hidden.** Only 20 of 80 sampled functions (25%) contained an eligible +adjacent, independent, simple-statement pair by our conservative heuristic (excludes calls, subscripts, +member targets, and any pair whose right-hand sides cross-reference the other's left-hand side). This is a +coverage limit of the *harness*, not a claim about the lens — most real functions in this codebase either +have fewer than two adjacent simple statements or have dependencies between them, and a looser heuristic +risks constructing a swap that is not actually independent. + +Re-run: `bench/readability_self_consistency.py --samples 80 --seed 20260918 --out consistency.tsv` +(deterministic for a fixed seed, corpus and binary — the sampler's own RNG is seeded, and the rename target +is chosen by frequency-then-lexical order rather than randomly, so re-running with the same seed reproduces +the same mutations). + +## 4. Precedent: we have already withdrawn a lens that failed exactly this kind of check + +This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of +`docs/LINEAGE.md`, and the top of `src/naminglens.h` itself, record `naming-body-mismatch` — a rule that +flagged a name whose tokens shared zero vocabulary with its own body. Measured on this repository's own +`src/` at the commit that shipped it, the rule produced 159 of the lens's 217 naming findings (73% of the +whole signal), and the flagged set was **dominated by the best-named functions in the tree** +(`didYouMean`, `transitiveCallers`, `symbolAdjacency`) — because a good abstraction name states *intent* +while its body states *mechanism*, so near-zero overlap is the signature of a successful abstraction at least +as often as of a lying name. The axis was non-monotonic with quality: no threshold on it has a defensible +direction. It was **withdrawn before it shipped**, and `naminglens.h` carries a do-not-re-add note plus the +measured numbers at the top of the file rather than a silent deletion. + +We take that as the template for what "failing this validation" would require of us, not just for naming: if +§2's human-rated pairwise study comes back at or below chance, or flips sign across a second corpus, the +correct response is the same one `naming-body-mismatch` got — record the numbers and the reasoning where the +next reader will actually see them (this document and `--readability`'s own header comment, the way +`naminglens.h`'s header comment carries its own withdrawal), demote or remove the claim from any joined +surface (`--ensemble`'s `rrank=` first), and do **not** quietly re-add a close cousin of it later without +citing why this round's finding no longer applies. §3's numbers are not at that bar yet — proxy (a)'s +70%-wrong-direction result on refactor commits is concerning enough that we think the human-rated study in +§2 is now the right next step, not an optional nice-to-have. + +## 5. What we would like help with + +We are not readability researchers; we are reporting what a deterministic, disclosed, ordering-only lens +measures against a construct it was never claimed to solve, and we would like informed pushback on the +following, specifically: + +1. **Is §2's proposed pairwise + rank-correlation protocol the right design**, or is there a better-controlled + study shape for validating an *ordering* claim (as opposed to the score-based designs most classic + readability datasets — Buse & Weimer, Scalabrino, Dorn — were built for)? We deliberately avoided asking + for a 1–5 scale per snippet, on Vitale's (2025) finding that a meaningful fraction of such labels are + self-contradictory on reread; is a forced pairwise choice actually more reliable, or does it just move the + inconsistency somewhere this document has not thought to look? +2. **What is a defensible pass/fail bar for an ordering-only metric**, given that the field's own consensus + (Scalabrino ASE'17, Trockman MSR'18) is that no classic metric correlates *strongly* with measured + understandability? §2 proposes three bands calibrated against "does the lens earn the narrow claim it + makes," not against "strong correlation is achievable" — is that the right frame, or does it let a weak + metric off too easily? +3. **Proxy (a)'s 70%-wrong-direction number** (§3a) is the most actionable finding in this document, and we + would like a sanity check on the method before we act on it: is commit-message mining (refactor/simplify/ + cleanup) too noisy a readability label on its own — conflating "the author changed something for reasons + unrelated to readability" with "the author made it more readable" — and if so, what filter (a stricter + subject regex, a manual pass over a sample, restricting to commits that touch exactly one function) would + make the label trustworthy enough to report a headline number from? +4. **Is Halstead volume, entropy and length the right feature set to be checking at all** in 2026, or is this + entire investigation validating a formula the field has already moved past? `DESIGN_READABILITY_METRICS.md` + §5 in this repo surveys later models (Buse & Weimer 2010, Scalabrino 2018) that add lexical/visual/textual + features on top of the same structural core — if there is a more recent, still-deterministic (no model + call) formula with better-established construct validity, we would rather adopt it than keep defending + Posnett 2011 out of inertia. + +## Reference list + +- Posnett, D., Hindle, A. & Devanbu, P. *A Simpler Model of Software Readability.* MSR 2011. + [doi:10.1145/1985441.1985454](https://doi.org/10.1145/1985441.1985454) +- Halstead, M. H. *Elements of Software Science.* Elsevier, 1977. +- Scalabrino, S., Linares-Vásquez, M., Poshyvanyk, D. & Oliveto, R. *Improving Code Readability Models with + Textual Features.* ICPC 2016 / *A Comprehensive Model for Code Readability.* JSEP 2018. + [doi:10.1109/ICPC.2016.7503707](https://doi.org/10.1109/ICPC.2016.7503707) +- Trockman, A. et al. — the MSR 2018 result cited throughout this repo's readability code as finding no + readability metric or combination correlates strongly with measured understandability (see + `src/readability.h`, `DESIGN_READABILITY_METRICS.md` §0). +- Fakhoury, S. et al. *Improving Source Code Readability: Theory and Practice.* ICPC 2019 — 548 + developer-declared readability-improving commits across 63 projects; classic models "fail to capture + readability improvements." +- Peitek, N., Apel, S., Parnin, C., Brechmann, A. & Siegmund, J. *Program Comprehension and Code Complexity + Metrics: An fMRI Study.* ICSE 2021. [doi:10.1109/ICSE43902.2021.00056](https://doi.org/10.1109/ICSE43902.2021.00056) — + Halstead volume specifically tracks measured cognitive load. +- Vitale, T. et al. (2025) — cited in `DESIGN_READABILITY_METRICS.md` §0 as finding up to a third of classic + readability ground-truth labels self-contradictory. +- The "Readability Spectrum" prompt-style-constraint study, arXiv:2605.13280 — style constraints in a + generation prompt help but plateau; cited in `DESIGN_READABILITY_METRICS.md` §7. +- CoReEval, arXiv:2510.16579 — LLM self-judges of code readability fixate on surface features; cited + alongside the prompt-constraint study in `DESIGN_READABILITY_METRICS.md` §7 as the joint reason + `--quality-delta`'s gate is deterministic and external to the model, applied to the diff. +- `docs/LINEAGE.md` §9.0 and `src/naminglens.h` (top-of-file comment) — the withdrawn `naming-body-mismatch` + rule, this repo's only other instance of "measured, then withdrawn," and the template §4 of this document + follows. From dc384e2acfd529fafb2d9cc6ea3761e2dce32a8d Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Mon, 21 Sep 2026 11:49:09 -0400 Subject: [PATCH 02/10] chore(privacy): drop the internal design-note filename, add the research row MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ripwirepubliccheck arm 8 fired: the reference list cited an internal working document by name eight times, and that document is not in this repository, so every one of those citations is a dangling link for any reader outside it. The citations stay — they are where the claims come from — spelled as "the readability design note" instead of a filename nobody else can open. docs/README.md gains the canonical `research/` row arm 6b requires, byte-identical to the one on every other research lane. Co-Authored-By: Claude Opus 5 --- docs/README.md | 3 ++- .../readability-construct-validity.md | 19 ++++++++++--------- 2 files changed, 12 insertions(+), 10 deletions(-) diff --git a/docs/README.md b/docs/README.md index a9ba460b8..4d459a24f 100644 --- a/docs/README.md +++ b/docs/README.md @@ -1,6 +1,6 @@ # ripwire documentation -Sixteen entries, each written for one reader. Start with the row that matches why you are here. +Twenty-one entries, each written for one reader. Start with the row that matches why you are here. | File | Who it is for | What it answers | | --- | --- | --- | @@ -22,6 +22,7 @@ Sixteen entries, each written for one reader. Start with the row that matches wh | **[`gatecount_build.py`](gatecount_build.py)** | Maintainers, and anyone adding a gate | The generator behind the **published gate count**. Derives it from the single `for _g in …; do` loop in `test/regression.sh` and rewrites all eight marked sites across `README.md`, `EVALS.md` and the deck; `--check` is the drift comparison that `test/gatecountcheck.sh` runs. Never edit that number by hand — two lanes hand-writing the same N+1 auto-merge clean against a loop of N+2. | | **[`limits_classes.tsv`](limits_classes.tsv)** | Maintainers | Cap name -> INDEXING or OUTPUT, the taxonomy `LIMITS.md` renders in its `class` column: does this cap bound what can EVER be found, or only what is shown from what was found. A sidecar with a known expiry — the tag belongs on the declaration in `src/` — kept honest by `limitstablecheck.sh`, which fails if a row names a cap that no longer exists. | | **[`lineage-paper-dates.tsv`](lineage-paper-dates.tsv)** | Maintainers | arXiv id -> publication date for every 2026 paper in `LINEAGE.md`. The ID stem does not track the date (`2607.09691` was published 2026-06-19), so the README's recency claim is re-derived from this file by `readmedriftcheck.sh` arm (H2) rather than from the ids. Adding a 2026 paper without a date row fails that arm. | +| **[`research/`](research/)** | Anyone working on an open question here, or offering to help with one | Investigation notes: a question this tool has not answered, the measurement run against it, and the pre-registration that says what would license a change. Not a claim surface — a number here is local evidence for a decision not yet taken, and nothing in it is quoted on `README.md`. | | **[`assets/`](assets/)** | The front page | The README banner and tagline artwork (SVG, self-contained). | | **[`captures/`](captures/)** | Maintainers, and the curious | One recorded run of every verb against a real repository — the source of `COMMANDS.md`'s sample output, and the harvest source for the differential argv harness. | diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index bec5163c1..f80fa3499 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -18,7 +18,8 @@ implementations** — nothing in `src/readability.h` runs a model or a judge: the actual diff, rather than living inside a generation prompt as one more instruction competing with the rest of the prompt for effect. -Both are cited in this repo's own research log, `DESIGN_READABILITY_METRICS.md` §7, as the "Readability +Both are cited in the readability design note — an internal working document, not part of this +repository — at its §7, as the "Readability Spectrum" prompt-constraint finding and as "LLM self-judges … fixate on surface features (CoReEval)" — see the reference list at the end of this document for both arXiv identifiers. That design log already states the honesty bound this document expands on: *"Everything here is a lens, never a verdict."* This document is @@ -93,7 +94,7 @@ reader would recognize. Here is the validation we think would settle that, speci **Unit of comparison: pairs, not scores.** Ask a rater "which of these two is more readable," never "rate this snippet 1–5" — the literature's own ground-truth datasets are pairwise-derived or scale-based and Vitale -(2025, cited in `DESIGN_READABILITY_METRICS.md` §7) found up to a third of classic scale labels +(2025, cited in the readability design note §7) found up to a third of classic scale labels self-contradictory on reread. A forced pairwise choice is cheaper to collect, cheaper to get consistent, and is exactly the shape `--readability`'s own claim needs checking against: it emits an order, so validate the order. @@ -118,7 +119,7 @@ order. `posnett=`, no commit message); each pair rated independently by at least three raters, majority vote as the pair's label, inter-rater agreement reported alongside the correlation (a low agreement number is itself a finding, per Vitale). Raters should not be told which side the lens preferred — Piantadosi et al.'s "readable -state flip" framing (cited in `DESIGN_READABILITY_METRICS.md` §6) is a reasonable model for how to phrase the +state flip" framing (cited in the readability design note §6) is a reasonable model for how to phrase the question without anchoring the rater on a metric. **The statistic.** Percentage pairwise agreement between the lens's implied direction and the majority human @@ -303,8 +304,8 @@ following, specifically: subject regex, a manual pass over a sample, restricting to commits that touch exactly one function) would make the label trustworthy enough to report a headline number from? 4. **Is Halstead volume, entropy and length the right feature set to be checking at all** in 2026, or is this - entire investigation validating a formula the field has already moved past? `DESIGN_READABILITY_METRICS.md` - §5 in this repo surveys later models (Buse & Weimer 2010, Scalabrino 2018) that add lexical/visual/textual + entire investigation validating a formula the field has already moved past? The readability design note + §5 surveys later models (Buse & Weimer 2010, Scalabrino 2018) that add lexical/visual/textual features on top of the same structural core — if there is a more recent, still-deterministic (no model call) formula with better-established construct validity, we would rather adopt it than keep defending Posnett 2011 out of inertia. @@ -319,19 +320,19 @@ following, specifically: [doi:10.1109/ICPC.2016.7503707](https://doi.org/10.1109/ICPC.2016.7503707) - Trockman, A. et al. — the MSR 2018 result cited throughout this repo's readability code as finding no readability metric or combination correlates strongly with measured understandability (see - `src/readability.h`, `DESIGN_READABILITY_METRICS.md` §0). + `src/readability.h`, the readability design note §0). - Fakhoury, S. et al. *Improving Source Code Readability: Theory and Practice.* ICPC 2019 — 548 developer-declared readability-improving commits across 63 projects; classic models "fail to capture readability improvements." - Peitek, N., Apel, S., Parnin, C., Brechmann, A. & Siegmund, J. *Program Comprehension and Code Complexity Metrics: An fMRI Study.* ICSE 2021. [doi:10.1109/ICSE43902.2021.00056](https://doi.org/10.1109/ICSE43902.2021.00056) — Halstead volume specifically tracks measured cognitive load. -- Vitale, T. et al. (2025) — cited in `DESIGN_READABILITY_METRICS.md` §0 as finding up to a third of classic +- Vitale, T. et al. (2025) — cited in the readability design note §0 as finding up to a third of classic readability ground-truth labels self-contradictory. - The "Readability Spectrum" prompt-style-constraint study, arXiv:2605.13280 — style constraints in a - generation prompt help but plateau; cited in `DESIGN_READABILITY_METRICS.md` §7. + generation prompt help but plateau; cited in the readability design note §7. - CoReEval, arXiv:2510.16579 — LLM self-judges of code readability fixate on surface features; cited - alongside the prompt-constraint study in `DESIGN_READABILITY_METRICS.md` §7 as the joint reason + alongside the prompt-constraint study in the readability design note §7 as the joint reason `--quality-delta`'s gate is deterministic and external to the model, applied to the diff. - `docs/LINEAGE.md` §9.0 and `src/naminglens.h` (top-of-file comment) — the withdrawn `naming-body-mismatch` rule, this repo's only other instance of "measured, then withdrawn," and the template §4 of this document From 867c84bcb759be6ab232e7c1a3d43626a804174d Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 06:35:20 -0400 Subject: [PATCH 03/10] =?UTF-8?q?research(readability):=20pre-register=20p?= =?UTF-8?q?roxy=20(c)=20and=20the=20=C2=A73a=20decomposition=20before=20co?= =?UTF-8?q?mputing=20them?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fixes the selection regex, the lens-name mask, the n target (100 directional pairs), and the decision bands (>=60% holds, <=40% inverted -> stop rule, between inconclusive) in the note before the harness exists or runs. Co-Authored-By: Claude Opus 5 --- .../readability-construct-validity.md | 72 +++++++++++++++++++ 1 file changed, 72 insertions(+) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index f80fa3499..b1de64a77 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -257,6 +257,78 @@ Re-run: `bench/readability_self_consistency.py --samples 80 --seed 20260918 --ou is chosen by frequency-then-lexical order rather than randomly, so re-running with the same seed reproduces the same mutations). +### 3c. A second, independent proxy: commits that DECLARE a readability improvement + +§3a leaves two explanations standing, and they predict different things: + +- **Lens defect** — the formula genuinely disagrees with readability. Predicts the inversion reappears on + *any* readability ground truth, including commits whose author says in so many words that the change is + for readability. +- **Proxy defect** — a "refactor" commit is not a readability improvement (refactors add guards, split + responsibilities, keep contracts), which is also the published result this lens already cites (Fakhoury + ICPC'19: classic models miss developer-declared readability improvements). Predicts that the §3a inversion + is explained by a mechanical component of the formula responding to non-readability edits, and says nothing + in advance about declared-readability commits. + +The stop rule this section runs under: **if a second, independent proxy also comes out inverted, we withdraw +the ordering claim rather than re-fit the formula to rescue it.** + +#### Protocol (fixed in this commit, before any number below was computed) + +This protocol was committed to this document before the harness that computes it was run; the commit that +adds this subsection predates the commit that adds its results, and nothing here was edited afterwards. + +**Corpus and binary.** Exactly §3a's: this repository's own git history, walked with the same +`git log --all` over `src/` C/C++ sources, the same ≤400 total-changed-lines cap, the same single-file +scratch scoring through the shipped `--readability` verb, the same (basename, name) matching, the same +"changed token shape" filter and the same full-precision pre-sigmoid `z` comparison. The binary is a v0.6.2 +build; `src/readability.h` is byte-identical between that build's source and this document's base. + +**Selection rule — proxy (c), primary arm.** A commit is selected when its **subject line**, after the lens's +own name is masked, matches the case-insensitive regex + +``` +readab|legib|clarity|clarif|clearer|easier to (read|follow|understand)|self-documenting +``` + +The mask replaces these substrings with a blank before matching, so a commit *about the lens* is not read as +a commit *declaring readability*: `--readability`, `readability.h`, `readability_`, and a conventional-commit +scope `(readability)` / `(readability,` / `readability)`. A subject that **also** matches §3a's refactor regex +(`refactor|simplify|clean(\s|-)?up`) is excluded, so the primary arm's commit set is disjoint from §3a's by +construction. No `--max-commits` cap: every qualifying commit in the history is used. + +**Secondary arm (descriptive only, no verdict).** The same rule applied to the full message (subject and +body), still masked and still disjoint from §3a. Bodies in this repository are long and mention readability in +passing, so this arm is noisier; it is reported because it was named here, and it carries no decision. + +**n target.** **100 directional function pairs** in the primary arm (at p = 0.5, 100 pairs gives a 95% +interval of about ±10 points, which is the width of the bands below). If the primary arm reaches fewer than +100, the result is reported as "n not reached", with the n that was reached and its rate shown for the record, +and **no verdict is drawn**. The regex is not loosened after looking. + +**Decision bands** (on the primary arm's fraction of directional pairs the lens ranks AFTER more readable): + +- **≥ 60%** → the lens's ordering holds on this proxy; §3a's inversion is read as a proxy defect. +- **≤ 40%** → inverted on a second, independent proxy; the stop rule triggers and this document recommends + withdrawing the ordering claim. +- **between** → inconclusive, reported as such; neither explanation is favoured by proxy (c). + +Because pairs cluster inside commits, the per-commit rate (a commit counts as "right" when most of its +directional pairs moved up) is also reported. It is secondary and does not move the band. + +**§3a decomposition (deterministic, fixed here too).** For each §3a pair, Δz splits exactly into three terms: +`ΔV·(−0.033)` (Halstead volume), `ΔE·(−1.5)` (entropy), `ΔL·(+0.40)` (lines). For every wrong-direction pair +(Δz < 0), the **driver** is the term with the most negative contribution. Reported: the driver split, the +mean contribution of each term over the wrong-direction pairs, and how many wrong-direction pairs got *longer* +(ΔL > 0). The same split over §3a's right-direction pairs is reported beside it for contrast. The pairs are +regenerated by re-running §3a's exact command (`--max-commits 80 --max-changed-lines 400`); because +`git log --all` sees every ref, a later ref set can shift which 80 commits are newest, so the regenerated +counts are reported beside the published 484 / 146 rather than assumed equal to them. + +No LLM judgment is used anywhere in this section — not for labels, not for spot-checks. The reference list +cites CoReEval for why: an LLM readability judge fixates on surface features, which is the very construct +question under test. + ## 4. Precedent: we have already withdrawn a lens that failed exactly this kind of check This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of From ba1c34299e9a4ce180e00a43f84914614debb62f Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 06:41:36 -0400 Subject: [PATCH 04/10] =?UTF-8?q?research(readability):=20proxy=20(c)=20an?= =?UTF-8?q?d=20the=20=C2=A73a=20decomposition=20=E2=80=94=20results?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Proxy (c) (declared-readability commits, pre-registered in 867c84bc) reached n=4 of the 100 target, all regex false positives: no verdict, the stop rule does not trigger. §3a regenerated at 413 pairs (38.3% right-direction); 91% of wrong-direction pairs are driven by the Halstead-volume term, and the sign of the token-count change predicts the lens's direction on 94.2% of pairs. No src/ change; a legend narrowing is proposed for owner sign-off only. Co-Authored-By: Claude Opus 5 --- bench/readability_declared_pairs.py | 182 ++++++++++++++++++ .../readability-construct-validity.md | 75 ++++++++ 2 files changed, 257 insertions(+) create mode 100755 bench/readability_declared_pairs.py diff --git a/bench/readability_declared_pairs.py b/bench/readability_declared_pairs.py new file mode 100755 index 000000000..04f2ded7b --- /dev/null +++ b/bench/readability_declared_pairs.py @@ -0,0 +1,182 @@ +#!/usr/bin/env python3 +r"""readability_declared_pairs.py — proxy (c) and the §3a decomposition of +docs/research/readability-construct-validity.md. + +Proxy (c): commits whose message DECLARES a readability improvement (the Fakhoury ICPC'19 style of ground +truth), scored with exactly §3a's pairing and z comparison (bench/readability_refactor_pairs.py is imported, +not copied, so the two proxies cannot drift apart). The selection rule below was fixed in the paired +document's §3c before this script existed; do not edit it without also saying so there. + +Decomposition: re-runs §3a's own selection and splits every pair's delta-z into its three exact terms — +Halstead volume, token entropy, lines — naming the most negative one as the driver of a wrong-direction pair. + +No LLM judgment anywhere: labels come from commit messages, directions from the shipped binary. + +A measurement harness is a LEDGER, never a red CI gate (bench/ANSWERQUALITY.md): exits 0 whatever it finds. + +Usage: + bench/readability_declared_pairs.py --bin build/ripwire # proxy (c), both arms + bench/readability_declared_pairs.py --bin build/ripwire --decompose # §3a decomposition +""" + +from __future__ import annotations + +import argparse +import re +import shutil +import sys +import tempfile +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import readability_refactor_pairs as rp # noqa: E402 + +# Fixed in §3c's protocol, before any result was computed. +DECLARED_RE = re.compile(r"readab|legib|clarity|clarif|clearer|easier to (read|follow|understand)|self-documenting", re.IGNORECASE) +LENS_NAME_MASK = re.compile(r"--readability|readability\.h|readability_|\(readability\)|\(readability,|readability\)", re.IGNORECASE) + + +def declares_readability(text: str) -> bool: + return DECLARED_RE.search(LENS_NAME_MASK.sub(" ", text)) is not None + + +def changed_lines(sha: str) -> int: + stat = rp.git("show", "--shortstat", "--format=", sha) + return sum(int(tok) for tok in re.findall(r"(\d+) (?:insertion|deletion)s?\(\+?-?\)?", stat)) + + +def declared_commits(max_changed_lines: int, use_body: bool) -> list[tuple[str, str]]: + raw = rp.git("log", "--format=%H\x1f%s\x1f%b\x1e", "--all", "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc") + out: list[tuple[str, str]] = [] + for rec in raw.split("\x1e"): + rec = rec.strip("\n") + if not rec.strip(): + continue + sha, _, rest = rec.partition("\x1f") + subject, _, body = rest.partition("\x1f") + if rp.REFACTOR_RE.search(subject): + continue # disjoint from §3a by construction + if not declares_readability(subject + ("\n" + body if use_body else "")): + continue + n = changed_lines(sha) + if n == 0 or n > max_changed_lines: + continue + out.append((sha, subject)) + return out + + +def pairs_for(binary: str, commits: list[tuple[str, str]], scratch: Path) -> list[rp.FnPair]: + pairs: list[rp.FnPair] = [] + for sha, subject in commits: + for path in rp.touched_source_files(sha): + before_text = rp.show_file(f"{sha}^", path) + after_text = rp.show_file(sha, path) + if before_text is None or after_text is None: + continue + basename = Path(path).name + try: + before_rows = rp.score_file(binary, scratch / "before", basename, before_text) + after_rows = rp.score_file(binary, scratch / "after", basename, after_text) + except RuntimeError as exc: + print(f"# skip {sha[:10]} {path}: {exc}", file=sys.stderr) + continue + for name, (lb, tb, volb, entb, pb) in before_rows.items(): + if name not in after_rows: + continue + la, ta, vola, enta, pa = after_rows[name] + if (lb, tb) == (la, ta): + continue + pairs.append(rp.FnPair(sha, subject, path, name, lb, la, tb, ta, volb, vola, entb, enta, pb, pa)) + return pairs + + +def report_direction(label: str, commits: list[tuple[str, str]], pairs: list[rp.FnPair]) -> None: + s = rp.summarize(pairs) + directional = s["improved"] + s["worsened"] + by_commit: dict[str, list[rp.FnPair]] = {} + for p in pairs: + by_commit.setdefault(p.sha, []).append(p) + right = sum(1 for ps in by_commit.values() if sum(p.improved for p in ps) > sum(p.worsened for p in ps)) + wrong = sum(1 for ps in by_commit.values() if sum(p.improved for p in ps) < sum(p.worsened for p in ps)) + deltas = sorted(p.delta for p in pairs) + median = deltas[len(deltas) // 2] if deltas else float("nan") + print(f"== {label}") + print(f"selected commits : {len(commits)} ({len(by_commit)} contributed >=1 pair)") + print(f"directional pairs : {directional} (improved {s['improved']}, worsened {s['worsened']}, tied {s['tied_within_1e-9']})") + frac = s["frac_improved_of_directional"] + print(f"fraction AFTER ranked more readable : {frac:.3f}" if frac == frac else "fraction: n/a") + print(f"per-commit majority right/wrong/split: {right}/{wrong}/{len(by_commit) - right - wrong}") + print(f"mean / median delta-z : {s['mean_z_delta']:.3f} / {median:.3f}") + for sha, subject in commits: + ps = by_commit.get(sha, []) + print(f" {sha[:10]} +{sum(p.improved for p in ps)} -{sum(p.worsened for p in ps)} {subject}") + + +TERMS = ("volume", "entropy", "lines") + + +def contributions(p: rp.FnPair) -> dict[str, float]: + return { + "volume": rp.POSNETT_VOLUME * (p.vol_after - p.vol_before), + "entropy": rp.POSNETT_ENTROPY * (p.ent_after - p.ent_before), + "lines": rp.POSNETT_LINES * (p.lines_after - p.lines_before), + } + + +def report_decomposition(pairs: list[rp.FnPair]) -> None: + for label, subset, pick in (("wrong-direction (delta-z < 0)", [p for p in pairs if p.worsened], min), + ("right-direction (delta-z > 0)", [p for p in pairs if p.improved], max)): + print(f"== {label}: {len(subset)} pairs") + if not subset: + continue + drivers = {t: 0 for t in TERMS} + sums = {t: 0.0 for t in TERMS} + for p in subset: + c = contributions(p) + drivers[pick(TERMS, key=lambda t: c[t])] += 1 + for t in TERMS: + sums[t] += c[t] + for t in TERMS: + print(f" driver={t:<8}: {drivers[t]:>4} ({drivers[t] / len(subset):.1%}) mean contribution {sums[t] / len(subset):+.3f}") + longer = sum(1 for p in subset if p.lines_after > p.lines_before) + shorter = sum(1 for p in subset if p.lines_after < p.lines_before) + more_toks = sum(1 for p in subset if p.toks_after > p.toks_before) + print(f" got longer (dL>0): {longer} shorter: {shorter} same length: {len(subset) - longer - shorter} more tokens: {more_toks}") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=rp.DEFAULT_BIN) + ap.add_argument("--max-changed-lines", type=int, default=400) + ap.add_argument("--decompose", action="store_true", help="decompose §3a's pairs instead of running proxy (c)") + ap.add_argument("--max-commits", type=int, default=80, help="§3a's commit cap, used only with --decompose") + ap.add_argument("--until", default=None, help="only commits committed at or before this date (git log --until); pins §3a's " + "newest-80 window against refs added later") + args = ap.parse_args() + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin}", file=sys.stderr) + return 2 + if args.until: + plain_git = rp.git + rp.git = lambda *a: plain_git(*a[:1], f"--until={args.until}", *a[1:]) if a and a[0] == "log" else plain_git(*a) + scratch = Path(tempfile.mkdtemp(prefix="rw_declpairs_")) + try: + if args.decompose: + pairs = rp.collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch) + s = rp.summarize(pairs) + print(f"§3a regenerated: {s['pairs']} pairs, improved {s['improved']}, worsened {s['worsened']}, tied {s['tied_within_1e-9']}") + report_decomposition(pairs) + else: + for label, use_body in (("PRIMARY arm (subject line)", False), ("SECONDARY arm (subject + body, descriptive only)", True)): + commits = declared_commits(args.max_changed_lines, use_body) + pairs = pairs_for(args.bin, commits, scratch) + report_direction(label, commits, pairs) + if not use_body: + report_decomposition(pairs) + finally: + shutil.rmtree(scratch, ignore_errors=True) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index b1de64a77..97af4a793 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -329,6 +329,81 @@ No LLM judgment is used anywhere in this section — not for labels, not for spo cites CoReEval for why: an LLM readability judge fixates on surface features, which is the very construct question under test. +#### Results (computed after the protocol above was committed and pushed) + +**Proxy (c), primary arm: n not reached — no verdict.** The pre-registered subject-line rule selected **4 +commits yielding 4 directional pairs** against a target of 100 (3 ranked more readable after, 1 less; 3/1 per +commit). Under the protocol that is reported as "n not reached" and draws no band. It is also worse than +small: **none of the four is a readability declaration.** Two match `readab` inside *unreadable* (commits +about files the tool could not read), and two match the lens's own name in a form the mask did not anticipate +(`readability-wave1`, a lane name). The mask gap is a defect of the pre-registered rule, disclosed here and not +patched after the fact. The honest summary: **this repository's `src/` history holds no commit whose subject +declares a readability improvement**, so proxy (c) cannot be run on this corpus at all. + +**Secondary arm: descriptive only, and not a second proxy.** Subject and body together selected 58 commits and +153 directional pairs, of which the lens ranked AFTER more readable on 43 (28.1%; 16/31/3 per commit +right/wrong/split). This arm carries no verdict by the protocol, and reading its commit list shows why it +should not: its subjects are overwhelmingly error-handling fixes (a file that "could not be read", a +directory walk that was "unreadable"), matched through words like *unreadable* in the body. It is a set of +bug-fix commits, not a readability ground truth, and a 28% figure on it says nothing about the lens-defect +explanation. + +**§3a decomposition.** Re-running §3a's command did **not** reproduce the published 484 pairs: with the +same lens source it gives 409 pairs (157 up, 252 down, 38.4% right-direction), and 413 pairs (158 up, 255 +down, 38.3%) with the commit window pinned to the date §3a was committed (`--until`). The difference is in +which commits the `git log --all` walk sees, not in the lens; both regenerations stay on the inverted side of +40%, and the table below uses the pinned one. + +| over the 413 regenerated pairs | wrong-direction (255) | right-direction (158) | +|---|---|---| +| driver = Halstead volume term `ΔV·(−0.033)` | **232 (91.0%)** | 152 (96.2%) | +| driver = lines term `ΔL·(+0.40)` | 23 (9.0%) | 6 (3.8%) | +| driver = entropy term `ΔE·(−1.5)` | 0 | 0 | +| mean contribution: volume / entropy / lines | −3.04 / −0.06 / +0.18 | +20.63 / +0.20 / −4.89 | +| function gained tokens | 236 | 3 | +| function got longer in lines / shorter / same | 45 / 51 / 159 | 8 / 131 / 19 | + +Three things follow, all mechanical: + +1. **The inversion is a token-count signal, not a length signal.** In the Posnett fit the lines coefficient + is *positive* (+0.40): holding volume fixed, a longer function scores *more* readable. So line count cannot + be what pushed a pair the wrong way unless the function got shorter, and on average it pushed the other + way (+0.18). The Halstead-volume term drove 91% of the wrong-direction pairs; entropy drove none. +2. **Direction is almost entirely the sign of the token-count change.** On 389 of the 413 pairs (94.2%), the + lens's direction is predicted by whether the function lost tokens (ranked more readable) or gained them + (ranked less readable); 7 pairs kept the same token count. +3. **The typical wrong-direction pair is a small addition.** Median over the 255: +5 tokens, 0 lines, Δz + −1.21 — a guard, a check, or an extra argument inside an unchanged line span, in a commit whose subject + said "refactor", "simplify" or "clean up". + +**Which explanation the data favours.** The lens-defect explanation is **untested**, not refuted: its +distinguishing prediction needs a declared-readability ground truth, and this corpus does not contain one. +What the data does settle is the mechanism behind §3a's number, and that mechanism is the shape the +proxy-defect explanation predicts: the lens did exactly what its formula says — it ranked a function that +grew by a few tokens as less readable — on commits that mostly grew functions by a few tokens. Whether those +small additions made the code more readable is precisely what a commit subject containing "refactor" does +not tell us. §3a is therefore better read as *"on this history, `--readability`'s direction is the sign of +the token-count change"* than as *"the lens is wrong 70% of the time"* — and the first statement is a +narrower, checkable fact about the lens that does not depend on the label at all. + +**The stop rule did not trigger.** It needs a second independent proxy to come out inverted, and proxy (c) +produced no verdict. The shipped `--readability` flag and its help text are unchanged by this section. One +narrowing is supported by the decomposition regardless of how the construct question resolves, and falls in +§2's middle band ("disclose that the lens is known to track length more than it tracks anything +len-independent"), refined by what was measured — it tracks *token count*, not lines. As a **proposal for +owner sign-off, not a change made here**, the legend could add: *"On ripwire's own history the ORDER between +two versions of a function followed the sign of its token-count change in 94% of pairs; read a move as 'more +or fewer tokens', not as more or less readable."* + +**Next step.** Proxy (c) needs a corpus that actually contains declared-readability commits — Fakhoury et +al.'s 548-commit set is the natural one — scored with this same script by pointing it at that checkout, under +this same protocol (bands, n target, and the stop rule unchanged; the mask gap above fixed *before* that run +and disclosed as a change). + +Re-run: `bench/readability_declared_pairs.py --bin build/ripwire` (proxy (c), both arms) and +`bench/readability_declared_pairs.py --bin build/ripwire --decompose --until=2026-09-20T14:42:20-04:00` +(the pinned §3a decomposition; omit `--until` for the unpinned 409-pair run). + ## 4. Precedent: we have already withdrawn a lens that failed exactly this kind of check This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of From 9aecbc96b77af71c0f9b913546ec578aece98139 Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 06:55:27 -0400 Subject: [PATCH 05/10] =?UTF-8?q?fix(research):=20pin=20readability=20?= =?UTF-8?q?=C2=A73a/=C2=A73c=20to=20v0.6.2,=20not=20git=20log=20--all?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit bench/readability_refactor_pairs.py and bench/readability_declared_pairs.py walked `git log --all`, which sweeps every branch in this clone's shared .git (~291 worktree lanes at the time this was found) — so the published pair count moved whenever an unrelated lane was pushed, independent of the lens. A number that moves when an unrelated branch is pushed is not a measurement. Both scripts now default to walking exactly one immutable ref (the v0.6.2 tag, matching the scoring binary's own build commit), overridable with --ref; --until (added for the prior workaround) is kept for narrowing within whichever ref is walked. docs/research/readability-construct-validity.md's §3a headline is now this pinned run (412 pairs, 154 right-direction, 37.4%) with the original unpinned figure kept visible as a one-line "first recorded as" note; §3c documents the instrument defect and reports its proxy (c) and decomposition numbers on the same pinned population. The inversion headline still holds (37.4% < 40%), just less starkly than the unpinned 30.2% — every other citation of the old numbers in the note (70%, ~338, 94.2%, etc.) is updated to match. test/ripwirepubliccheck.sh: ALL PASS. Co-Authored-By: Claude Sonnet 5 --- bench/readability_declared_pairs.py | 10 +- bench/readability_refactor_pairs.py | 27 +++- .../readability-construct-validity.md | 152 +++++++++++------- 3 files changed, 117 insertions(+), 72 deletions(-) diff --git a/bench/readability_declared_pairs.py b/bench/readability_declared_pairs.py index 04f2ded7b..9aedc268f 100755 --- a/bench/readability_declared_pairs.py +++ b/bench/readability_declared_pairs.py @@ -45,8 +45,8 @@ def changed_lines(sha: str) -> int: return sum(int(tok) for tok in re.findall(r"(\d+) (?:insertion|deletion)s?\(\+?-?\)?", stat)) -def declared_commits(max_changed_lines: int, use_body: bool) -> list[tuple[str, str]]: - raw = rp.git("log", "--format=%H\x1f%s\x1f%b\x1e", "--all", "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc") +def declared_commits(max_changed_lines: int, use_body: bool, ref: str = rp.DEFAULT_REF) -> list[tuple[str, str]]: + raw = rp.git("log", "--format=%H\x1f%s\x1f%b\x1e", ref, "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc") out: list[tuple[str, str]] = [] for rec in raw.split("\x1e"): rec = rec.strip("\n") @@ -150,6 +150,8 @@ def main() -> int: ap.add_argument("--max-changed-lines", type=int, default=400) ap.add_argument("--decompose", action="store_true", help="decompose §3a's pairs instead of running proxy (c)") ap.add_argument("--max-commits", type=int, default=80, help="§3a's commit cap, used only with --decompose") + ap.add_argument("--ref", default=rp.DEFAULT_REF, + help=f"walk history from exactly this ref (default {rp.DEFAULT_REF!r}) — never --all") ap.add_argument("--until", default=None, help="only commits committed at or before this date (git log --until); pins §3a's " "newest-80 window against refs added later") args = ap.parse_args() @@ -162,13 +164,13 @@ def main() -> int: scratch = Path(tempfile.mkdtemp(prefix="rw_declpairs_")) try: if args.decompose: - pairs = rp.collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch) + pairs = rp.collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch, args.ref) s = rp.summarize(pairs) print(f"§3a regenerated: {s['pairs']} pairs, improved {s['improved']}, worsened {s['worsened']}, tied {s['tied_within_1e-9']}") report_decomposition(pairs) else: for label, use_body in (("PRIMARY arm (subject line)", False), ("SECONDARY arm (subject + body, descriptive only)", True)): - commits = declared_commits(args.max_changed_lines, use_body) + commits = declared_commits(args.max_changed_lines, use_body, args.ref) pairs = pairs_for(args.bin, commits, scratch) report_direction(label, commits, pairs) if not use_body: diff --git a/bench/readability_refactor_pairs.py b/bench/readability_refactor_pairs.py index e162fc2e9..f9cd0e647 100755 --- a/bench/readability_refactor_pairs.py +++ b/bench/readability_refactor_pairs.py @@ -49,6 +49,13 @@ ROOT = Path(__file__).resolve().parent.parent DEFAULT_BIN = os.environ.get("RIPWIRE_BIN", str(ROOT / "build" / "ripwire")) +# The population must be pinned to ONE immutable ref, never `--all`: this clone's shared .git carries every +# worktree's branches, so `--all` makes the candidate-commit population — and therefore the published +# fraction — move whenever any unrelated lane is pushed. That is an instrument defect, not a measurement +# (docs/research/readability-construct-validity.md §3a). v0.6.2 is the tag the shipped scoring binary was +# built from; override with --ref for a different pinned population, but never pass --all here. +DEFAULT_REF = "v0.6.2" + REFACTOR_RE = re.compile(r"refactor|simplify|clean(\s|-)?up", re.IGNORECASE) SOURCE_EXT = {".h", ".hpp", ".cc", ".cpp", ".cxx"} @@ -115,12 +122,14 @@ def git(*args: str) -> str: return proc.stdout -def candidate_commits(max_commits: int, max_changed_lines: int) -> list[tuple[str, str]]: +def candidate_commits(max_commits: int, max_changed_lines: int, ref: str = DEFAULT_REF) -> list[tuple[str, str]]: """(sha, subject) pairs, newest first, whose subject reads as a refactor/simplify/cleanup and whose total changed-line count over the whole commit is small enough that attributing a per-function score - delta to "the refactor" is defensible rather than noise from an unrelated bulk edit riding along.""" + delta to "the refactor" is defensible rather than noise from an unrelated bulk edit riding along. + + Walks exactly ONE immutable ref (default the v0.6.2 tag), never `--all`: see DEFAULT_REF's comment.""" raw = git( - "log", "--format=%H\x1f%s", "--all", "-i", + "log", "--format=%H\x1f%s", ref, "-i", "--grep=refactor", "--grep=simplify", "--grep=clean up", "--grep=cleanup", "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc", ) @@ -184,10 +193,10 @@ def score_file(binary: str, scratch_dir: Path, basename: str, text: str) -> dict return rows -def collect_pairs(binary: str, max_commits: int, max_changed_lines: int, scratch: Path) -> list[FnPair]: +def collect_pairs(binary: str, max_commits: int, max_changed_lines: int, scratch: Path, ref: str = DEFAULT_REF) -> list[FnPair]: pairs: list[FnPair] = [] - commits = candidate_commits(max_commits, max_changed_lines) - print(f"# {len(commits)} candidate refactor/simplify/cleanup commits (max_changed_lines={max_changed_lines})", file=sys.stderr) + commits = candidate_commits(max_commits, max_changed_lines, ref) + print(f"# {len(commits)} candidate refactor/simplify/cleanup commits (max_changed_lines={max_changed_lines}, ref={ref})", file=sys.stderr) before_dir = scratch / "before" after_dir = scratch / "after" for sha, subject in commits: @@ -237,6 +246,10 @@ def main() -> int: ap.add_argument("--max-changed-lines", type=int, default=400, help="skip commits whose total insertion+deletion count exceeds this (default 400) — " "keeps the compared functions attributable to the named refactor") + ap.add_argument("--ref", default=DEFAULT_REF, + help=f"walk history from exactly this ref (default {DEFAULT_REF!r}) — never --all, " + "which makes the population move whenever any unrelated branch is pushed to a " + "shared .git") ap.add_argument("--out", default=None, help="write the per-pair TSV here (default: stdout table only)") ap.add_argument("--json", action="store_true", help="print the summary as JSON instead of a table") ap.add_argument("--scratch", default=None, help="scratch directory (default: a fresh temp dir)") @@ -248,7 +261,7 @@ def main() -> int: scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_readpairs_")) try: - pairs = collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch) + pairs = collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch, args.ref) finally: if not args.scratch: shutil.rmtree(scratch, ignore_errors=True) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index 97af4a793..029d0f5de 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -169,35 +169,42 @@ saturates to a repeated `"0.000"` for most real functions (see the tie count bel falsely indistinguishable. `z` is monotonic in `posnett` (same sigmoid), so ranking by `z` ranks identically to ranking by `posnett` without the display truncation. -**Run** (`--max-commits 80 --max-changed-lines 400`, the numbers below): +**Run** (`--max-commits 80 --max-changed-lines 400`, history pinned to the `v0.6.2` tag — never `git log +--all`, the numbers below): | | | |---|---| -| candidate refactor/simplify/cleanup commits | 80 (78 contributed ≥1 usable pair) | -| function pairs (changed shape, matched by name) | 484 | -| lens ranks AFTER more readable (`z` increased) | 146 (30.2% of the 484 directional pairs) | -| lens ranks AFTER less readable (`z` decreased) | 338 (69.8%) | +| candidate refactor/simplify/cleanup commits | 80 (77 contributed ≥1 usable pair) | +| function pairs (changed shape, matched by name) | 412 | +| lens ranks AFTER more readable (`z` increased) | 154 (37.4% of the 412 directional pairs) | +| lens ranks AFTER less readable (`z` decreased) | 258 (62.6%) | | exact tie | 0 | -| mean Δz (after − before, + = more readable) | **+1.89** | -| median Δz | **−0.73** | -| pairs with \|Δz\| > 20 (large swings) | 27 / 484 (5.6%) | +| mean Δz (after − before, + = more readable) | **+3.79** | +| median Δz | **−0.56** | +| pairs with \|Δz\| > 20 (large swings) | 32 / 412 (7.8%) | -**Read this plainly, including that it is not flattering.** On a majority (70%) of the function pairs a human +First recorded as 484 pairs / 146 right-direction / 30.2% from an unpinned `git log --all` walk, which does +not reproduce; see §3c for the instrument defect, the fix, and the full decomposition. + +**Read this plainly, including that it is not flattering.** On a majority (63%) of the function pairs a human called a refactor, the lens's own ranking moved the *wrong* direction — it scored the after-version as *less* -readable. The mean is positive only because a small number of large swings (5.6% of pairs, all from commits +readable. The mean is positive only because a small number of large swings (7.8% of pairs, all from commits that genuinely shrank a function by splitting work into named helpers) pull it there; the median, which a skewed distribution like this one should be read against, is negative. This lines up with exactly the literature the lens's own code already cites as a reason for caution — Fakhoury ICPC'19's finding that classic readability models "miss real-world readability-improvement commits" is not a hypothetical risk here, it is what this run measured on ripwire's own history. -**Two worked examples, to show what is and is not driving the number.** The eight largest positive swings -are all commits whose stated purpose was extraction — moving a block of logic out into named helper -functions, which mechanically shortens the function the lens is scoring: `buildFieldNarrowTables` -(114→50 lines), `computeSnapshot` (116→61), `buildScopedRecvDecls` (61→15), `hasNetExfilShape` (59→17), -`buildExternalVetoTables` (147→86), `resolve` in `src/elixir_resolve.h` (57→17), `parseAsan` (99→21) and -`runSkipped` (71→45) — a 25%–75% line-count cut in every case. The lens's length-sensitivity is doing -exactly the intuitive thing there. The single +**Two worked examples, to show what is and is not driving the number.** The eight largest positive swings on +the pinned population are all commits whose stated purpose was extraction — moving a block of logic out into +named helper functions, which mechanically shortens (or hollows out) the function the lens is scoring: +`buildScipOverlay` (227→124 lines), `ingest` (1138→1029), `writePinCensus` (88→38), `buildFieldNarrowTables` +(114→50), `computeSnapshot` (116→61), `buildScopedRecvDecls` (61→15), `hasNetExfilShape` (59→17) and +`buildExternalVetoTables` (147→86) — a 41%–75% line-count cut in all but one case. The exception is `ingest` +(`refactor(ingest): the doc post-pass becomes ingest_docpass.h`): only a 9.6% line cut, but the swing is +still large because a token-dense block moved out into the new header — the Halstead-volume term collapsed +even though the line count barely moved. That is the same volume-not-lines mechanism §3c's decomposition +finds behind 91%+ of the wrong-direction pairs too (see below), not a second, unrelated effect. The single largest *negative* swing is the opposite shape of edit: commit `15af398e` (`refactor(L1-fix): keep existing contracts; readings spell no element markup`) inlined a one-line forwarding wrapper (`classify()`, 5 lines) into what had been a same-named sibling implementation, producing one 126-line @@ -209,7 +216,8 @@ the kind of case a pairwise human study (§2) would need to arbitrate, and we wo proxy if we quietly excluded it. Re-run: `bench/readability_refactor_pairs.py --max-commits 80 --out pairs.tsv` (defaults to this repo, -`build/ripwire`; deterministic given a fixed git history and a fixed binary). +`build/ripwire`, and the `v0.6.2` ref — override with `--ref`, never `--all`; deterministic given a fixed +git history and a fixed binary). ### 3b. Self-consistency under meaning-preserving rewrites (`bench/readability_self_consistency.py`) @@ -278,11 +286,12 @@ the ordering claim rather than re-fit the formula to rescue it.** This protocol was committed to this document before the harness that computes it was run; the commit that adds this subsection predates the commit that adds its results, and nothing here was edited afterwards. -**Corpus and binary.** Exactly §3a's: this repository's own git history, walked with the same -`git log --all` over `src/` C/C++ sources, the same ≤400 total-changed-lines cap, the same single-file -scratch scoring through the shipped `--readability` verb, the same (basename, name) matching, the same -"changed token shape" filter and the same full-precision pre-sigmoid `z` comparison. The binary is a v0.6.2 -build; `src/readability.h` is byte-identical between that build's source and this document's base. +**Corpus and binary.** Exactly §3a's: this repository's own git history, walked with the same pinned `v0.6.2` +ref over `src/` C/C++ sources (originally `git log --all` — see the instrument-fix note in Results below), +the same ≤400 total-changed-lines cap, the same single-file scratch scoring through the shipped +`--readability` verb, the same (basename, name) matching, the same "changed token shape" filter and the same +full-precision pre-sigmoid `z` comparison. The binary is a v0.6.2 build; `src/readability.h` is byte-identical +between that build's source and this document's base. **Selection rule — proxy (c), primary arm.** A commit is selected when its **subject line**, after the lens's own name is masked, matches the case-insensitive regex @@ -321,47 +330,67 @@ directional pairs moved up) is also reported. It is secondary and does not move (Δz < 0), the **driver** is the term with the most negative contribution. Reported: the driver split, the mean contribution of each term over the wrong-direction pairs, and how many wrong-direction pairs got *longer* (ΔL > 0). The same split over §3a's right-direction pairs is reported beside it for contrast. The pairs are -regenerated by re-running §3a's exact command (`--max-commits 80 --max-changed-lines 400`); because -`git log --all` sees every ref, a later ref set can shift which 80 commits are newest, so the regenerated -counts are reported beside the published 484 / 146 rather than assumed equal to them. +regenerated by re-running §3a's exact command (`--max-commits 80 --max-changed-lines 400`), now pinned to the +`v0.6.2` ref by default — walking `git log --all` instead would see every ref in this clone's shared `.git` +(~291 worktree branches at the time this defect was found), so a later ref set could silently shift which 80 +commits are "newest" and move the published fraction with it. That is exactly what happened to §3a's first +recorded number; see the instrument-fix note below. No LLM judgment is used anywhere in this section — not for labels, not for spot-checks. The reference list cites CoReEval for why: an LLM readability judge fixates on surface features, which is the very construct question under test. -#### Results (computed after the protocol above was committed and pushed) - -**Proxy (c), primary arm: n not reached — no verdict.** The pre-registered subject-line rule selected **4 -commits yielding 4 directional pairs** against a target of 100 (3 ranked more readable after, 1 less; 3/1 per +#### Instrument fix (this revision) — read this before the numbers below + +The first recorded run of this section (484 pairs / 146 right-direction / 30.2% in §3a; the proxy (c) and +decomposition numbers this subsection used to report) walked `git log --all` in both +`bench/readability_refactor_pairs.py` and `bench/readability_declared_pairs.py`. This clone's `.git` is +shared across every worktree in the orchestration tree that produced this document — at the time the defect +was found, that was on the order of 291 branches — so `--all` pulled in whatever those branches happened to +contain that day. A rerun of the *identical, unmodified* script gave 413 pairs (38.3%) with the commit window +pinned to the original run's date and 409 pairs (38.4%) without even that pin; neither matched the published +484/146/30.2%, and neither run had touched `src/readability.h` or the lens in any way. A number that moves +when an unrelated lane is pushed is not measuring the lens — it is measuring which branches exist in the +shared `.git` right now. That is an instrument defect, not a finding. + +Both scripts now default to walking exactly one immutable ref — the `v0.6.2` tag, which is also the ref the +scoring binary (`build/ripwire`, `built_from=15a20855c`) was built from — instead of `--all`. `--ref` +overrides it for a deliberately different population; `--until` (added when the defect was first worked +around) still narrows further within whatever ref is walked. §3a's headline table above is this pinned run. +Every number for the rest of this section is the same pinned run too, so — unlike the first version of this +document — the "published" and "regenerated" numbers here are one run, not two. + +#### Results (recomputed on the `v0.6.2`-pinned instrument) + +**Proxy (c), primary arm: n not reached — no verdict.** The pre-registered subject-line rule selected **3 +commits yielding 3 directional pairs** against a target of 100 (2 ranked more readable after, 1 less; 2/1 per commit). Under the protocol that is reported as "n not reached" and draws no band. It is also worse than -small: **none of the four is a readability declaration.** Two match `readab` inside *unreadable* (commits -about files the tool could not read), and two match the lens's own name in a form the mask did not anticipate -(`readability-wave1`, a lane name). The mask gap is a defect of the pre-registered rule, disclosed here and not -patched after the fact. The honest summary: **this repository's `src/` history holds no commit whose subject -declares a readability improvement**, so proxy (c) cannot be run on this corpus at all. - -**Secondary arm: descriptive only, and not a second proxy.** Subject and body together selected 58 commits and -153 directional pairs, of which the lens ranked AFTER more readable on 43 (28.1%; 16/31/3 per commit +small: **none of the three is a readability declaration.** Two match `readab` inside *unreadable* (commits +about files the tool could not read), and one matches the lens's own name in a form the mask did not +anticipate (`readability-wave1`, a lane name). The mask gap is a defect of the pre-registered rule, disclosed +here and not patched after the fact. The honest summary: **this repository's `src/` history, at `v0.6.2`, +holds no commit whose subject declares a readability improvement**, so proxy (c) cannot be run on this corpus +at all. + +**Secondary arm: descriptive only, and not a second proxy.** Subject and body together selected 46 commits +and 134 directional pairs, of which the lens ranked AFTER more readable on 38 (28.4%; 13/24/2 per commit right/wrong/split). This arm carries no verdict by the protocol, and reading its commit list shows why it should not: its subjects are overwhelmingly error-handling fixes (a file that "could not be read", a directory walk that was "unreadable"), matched through words like *unreadable* in the body. It is a set of bug-fix commits, not a readability ground truth, and a 28% figure on it says nothing about the lens-defect explanation. -**§3a decomposition.** Re-running §3a's command did **not** reproduce the published 484 pairs: with the -same lens source it gives 409 pairs (157 up, 252 down, 38.4% right-direction), and 413 pairs (158 up, 255 -down, 38.3%) with the commit window pinned to the date §3a was committed (`--until`). The difference is in -which commits the `git log --all` walk sees, not in the lens; both regenerations stay on the inverted side of -40%, and the table below uses the pinned one. +**§3a decomposition.** Re-running §3a's command against the pinned `v0.6.2` ref reproduces §3a's own headline +exactly: 412 pairs (154 up, 258 down, 37.4% right-direction). The table below is that same population. -| over the 413 regenerated pairs | wrong-direction (255) | right-direction (158) | +| over the 412 pinned pairs | wrong-direction (258) | right-direction (154) | |---|---|---| -| driver = Halstead volume term `ΔV·(−0.033)` | **232 (91.0%)** | 152 (96.2%) | -| driver = lines term `ΔL·(+0.40)` | 23 (9.0%) | 6 (3.8%) | +| driver = Halstead volume term `ΔV·(−0.033)` | **235 (91.1%)** | 148 (96.1%) | +| driver = lines term `ΔL·(+0.40)` | 23 (8.9%) | 6 (3.9%) | | driver = entropy term `ΔE·(−1.5)` | 0 | 0 | -| mean contribution: volume / entropy / lines | −3.04 / −0.06 / +0.18 | +20.63 / +0.20 / −4.89 | -| function gained tokens | 236 | 3 | -| function got longer in lines / shorter / same | 45 / 51 / 159 | 8 / 131 / 19 | +| mean contribution: volume / entropy / lines | −3.07 / −0.06 / +0.18 | +19.79 / +0.20 / −4.91 | +| function gained tokens | 239 | 3 | +| function got longer in lines / shorter / same | 47 / 51 / 160 | 8 / 127 / 19 | Three things follow, all mechanical: @@ -369,11 +398,11 @@ Three things follow, all mechanical: is *positive* (+0.40): holding volume fixed, a longer function scores *more* readable. So line count cannot be what pushed a pair the wrong way unless the function got shorter, and on average it pushed the other way (+0.18). The Halstead-volume term drove 91% of the wrong-direction pairs; entropy drove none. -2. **Direction is almost entirely the sign of the token-count change.** On 389 of the 413 pairs (94.2%), the - lens's direction is predicted by whether the function lost tokens (ranked more readable) or gained them - (ranked less readable); 7 pairs kept the same token count. -3. **The typical wrong-direction pair is a small addition.** Median over the 255: +5 tokens, 0 lines, Δz - −1.21 — a guard, a check, or an extra argument inside an unchanged line span, in a commit whose subject +2. **Direction is almost entirely the sign of the token-count change.** On 388 of the 404 pairs whose token + count actually changed (96.0%), the lens's direction is predicted by whether the function lost tokens + (ranked more readable) or gained them (ranked less readable); 8 of the 412 pairs kept the same token count. +3. **The typical wrong-direction pair is a small addition.** Median over the 258: +5 tokens, 0 lines, Δz + −1.26 — a guard, a check, or an extra argument inside an unchanged line span, in a commit whose subject said "refactor", "simplify" or "clean up". **Which explanation the data favours.** The lens-defect explanation is **untested**, not refuted: its @@ -383,7 +412,7 @@ proxy-defect explanation predicts: the lens did exactly what its formula says grew by a few tokens as less readable — on commits that mostly grew functions by a few tokens. Whether those small additions made the code more readable is precisely what a commit subject containing "refactor" does not tell us. §3a is therefore better read as *"on this history, `--readability`'s direction is the sign of -the token-count change"* than as *"the lens is wrong 70% of the time"* — and the first statement is a +the token-count change"* than as *"the lens is wrong 63% of the time"* — and the first statement is a narrower, checkable fact about the lens that does not depend on the label at all. **The stop rule did not trigger.** It needs a second independent proxy to come out inverted, and proxy (c) @@ -392,7 +421,7 @@ narrowing is supported by the decomposition regardless of how the construct ques §2's middle band ("disclose that the lens is known to track length more than it tracks anything len-independent"), refined by what was measured — it tracks *token count*, not lines. As a **proposal for owner sign-off, not a change made here**, the legend could add: *"On ripwire's own history the ORDER between -two versions of a function followed the sign of its token-count change in 94% of pairs; read a move as 'more +two versions of a function followed the sign of its token-count change in 96% of pairs; read a move as 'more or fewer tokens', not as more or less readable."* **Next step.** Proxy (c) needs a corpus that actually contains declared-readability commits — Fakhoury et @@ -400,9 +429,10 @@ al.'s 548-commit set is the natural one — scored with this same script by poin this same protocol (bands, n target, and the stop rule unchanged; the mask gap above fixed *before* that run and disclosed as a change). -Re-run: `bench/readability_declared_pairs.py --bin build/ripwire` (proxy (c), both arms) and -`bench/readability_declared_pairs.py --bin build/ripwire --decompose --until=2026-09-20T14:42:20-04:00` -(the pinned §3a decomposition; omit `--until` for the unpinned 409-pair run). +Re-run: `bench/readability_declared_pairs.py --bin build/ripwire` (proxy (c), both arms, `v0.6.2`-pinned by +default) and `bench/readability_declared_pairs.py --bin build/ripwire --decompose` (the §3a decomposition, on +the same pinned population as §3a itself). `--ref` points either command at a different immutable ref; +`--until` still narrows within whichever ref is walked. ## 4. Precedent: we have already withdrawn a lens that failed exactly this kind of check @@ -424,7 +454,7 @@ next reader will actually see them (this document and `--readability`'s own head `naminglens.h`'s header comment carries its own withdrawal), demote or remove the claim from any joined surface (`--ensemble`'s `rrank=` first), and do **not** quietly re-add a close cousin of it later without citing why this round's finding no longer applies. §3's numbers are not at that bar yet — proxy (a)'s -70%-wrong-direction result on refactor commits is concerning enough that we think the human-rated study in +63%-wrong-direction result on refactor commits is concerning enough that we think the human-rated study in §2 is now the right next step, not an optional nice-to-have. ## 5. What we would like help with @@ -444,7 +474,7 @@ following, specifically: understandability? §2 proposes three bands calibrated against "does the lens earn the narrow claim it makes," not against "strong correlation is achievable" — is that the right frame, or does it let a weak metric off too easily? -3. **Proxy (a)'s 70%-wrong-direction number** (§3a) is the most actionable finding in this document, and we +3. **Proxy (a)'s 63%-wrong-direction number** (§3a) is the most actionable finding in this document, and we would like a sanity check on the method before we act on it: is commit-message mining (refactor/simplify/ cleanup) too noisy a readability label on its own — conflating "the author changed something for reasons unrelated to readability" with "the author made it more readable" — and if so, what filter (a stricter From 6e6b5a896309de1d0cbde84b87eac4ff4dbfe8fc Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 09:57:02 -0400 Subject: [PATCH 06/10] fix(research): correct arXiv:2605.13280's title in the reference list The reference list called arXiv:2605.13280 the "Readability Spectrum" study -- a title the paper does not have. Its real title, from the arXiv export API, is "Characterizing Readability Issue Patterns and the Role of Prompt Design in LLM-Generated Code" (Ye, Ran, Xu, Zhou). Corrected all three occurrences (the "published results" intro, the design-note cross-reference, and the reference-list entry) and reworded each to state only what the paper's abstract supports: prompt design's overall role in generated-code readability is bounded. No sentence in the note now attributes to this paper a finding it does not contain. Co-Authored-By: Claude Sonnet 5 --- .../readability-construct-validity.md | 20 ++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index 029d0f5de..0a1610446 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -13,15 +13,16 @@ implementations** — nothing in `src/readability.h` runs a model or a judge: ripwire's readability lens is a deterministic, closed-form formula (Halstead volume, token entropy, the Posnett sigmoid fit) instead of a model call, and why the underlying evidence is disclosed (`vol=`, `ent=`, `posnett=`) rather than a single opaque number. -- A study of prompt-side style constraints on LLM code generation found they help but plateau — part of why - ripwire's readability-adjacent gate (`--quality-delta`'s `verbosity` kind) runs **after** each edit, against - the actual diff, rather than living inside a generation prompt as one more instruction competing with the - rest of the prompt for effect. +- A study of prompt design's association with LLM-generated code readability found the overall role of + prompt design bounded (Ye, Ran, Xu & Zhou, arXiv:2605.13280) — part of why ripwire's readability-adjacent + gate (`--quality-delta`'s `verbosity` kind) runs **after** each edit, against the actual diff, rather than + living inside a generation prompt as one more instruction competing with the rest of the prompt for + effect. Both are cited in the readability design note — an internal working document, not part of this -repository — at its §7, as the "Readability -Spectrum" prompt-constraint finding and as "LLM self-judges … fixate on surface features (CoReEval)" — -see the reference list at the end of this document for both arXiv identifiers. That design log already states +repository — at its §7, as the prompt-design-bounded finding (Ye et al., arXiv:2605.13280) and as "LLM +self-judges … fixate on surface features (CoReEval)" — see the reference list at the end of this document +for both arXiv identifiers. That design log already states the honesty bound this document expands on: *"Everything here is a lens, never a verdict."* This document is the follow-through on that bound — checking it empirically rather than repeating it. @@ -506,8 +507,9 @@ following, specifically: Halstead volume specifically tracks measured cognitive load. - Vitale, T. et al. (2025) — cited in the readability design note §0 as finding up to a third of classic readability ground-truth labels self-contradictory. -- The "Readability Spectrum" prompt-style-constraint study, arXiv:2605.13280 — style constraints in a - generation prompt help but plateau; cited in the readability design note §7. +- Ye, H., Ran, F., Xu, W. & Zhou, M. *Characterizing Readability Issue Patterns and the Role of Prompt + Design in LLM-Generated Code.* arXiv:2605.13280 — prompt design's role in generated-code readability is + bounded; cited in the readability design note §7. - CoReEval, arXiv:2510.16579 — LLM self-judges of code readability fixate on surface features; cited alongside the prompt-constraint study in the readability design note §7 as the joint reason `--quality-delta`'s gate is deterministic and external to the model, applied to the diff. From 61107ef643181fc2175f70a507b10ee069c7cbe7 Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 16:45:25 -0400 Subject: [PATCH 07/10] research(readability): pre-register a later-fix-rate validation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Neither §2 nor §3 tests the claim an agent actually relies on when it uses --readability to pick a refactor target: does a low score predict that a function needs fixing later? Adds bench/readability_fixrate_validity.py and this document's new §4 protocol, fixed before any number exists: population at a pinned v0.6.2-history cutoff (never --all), exposure = z (readability) and ccx (the same cognitive-complexity metric --quality-delta's "complexity" kind already trusts) read at the cutoff, outcome = touched by a fix-shaped commit in a pinned multi-week follow-up window (function spans re-resolved per commit so line drift cannot misattribute a hunk), statistic = risk ratio with Wilson/Katz CIs, raw and size-stratified. Discloses in writing, before running it, that a lines-based size stratification cannot fully rule out the token-volume confound §3c already found behind z's direction. No src/ change. Co-Authored-By: Claude Sonnet 5 --- bench/readability_fixrate_validity.py | 421 ++++++++++++++++++ .../readability-construct-validity.md | 83 +++- 2 files changed, 499 insertions(+), 5 deletions(-) create mode 100644 bench/readability_fixrate_validity.py diff --git a/bench/readability_fixrate_validity.py b/bench/readability_fixrate_validity.py new file mode 100644 index 000000000..7b2ec4c28 --- /dev/null +++ b/bench/readability_fixrate_validity.py @@ -0,0 +1,421 @@ +#!/usr/bin/env python3 +r"""readability_fixrate_validity.py — does `--readability`'s ordering predict anything worth acting on? + +§2 and §3 of docs/research/readability-construct-validity.md validate the ORDER against refactor-commit +direction and self-consistency. Neither asks the question an agent actually relies on when it uses the lens +to pick a refactor target: **does a low readability score predict that a function will need fixing later?** +This script answers that, with no human label and no LLM judge, against the shipped `--readability` lens and +a complexity control the repo's own `--quality-delta` "complexity" kind already trusts (cognitive complexity, +`ccx=`), on the SAME population, with and without controlling for function size. + +Protocol (fixed in this file before any number below existed; see the paired document's new section for the +narrative write-up): + + 1. POPULATION: every `fn`/`method` in `src/**/*.{h,hpp,cpp,cc}` at a pinned CUTOFF commit on `v0.6.2`'s + history (never `--all` — see the instrument-fix note in the paired doc's §3c). The cutoff is chosen to + leave a multi-week, thousand-plus-commit follow-up window to the pinned UNTIL ref (default `v0.6.2`) + so the outcome (see below) has room to happen. Population functions are matched between a real + `--readability` crawl and a real `--metrics` crawl of the SAME extracted tree (`git archive`, not + per-file scratch, so cross-file structure is intact for `--metrics`'s in/out/ccx) by + (path relative to root, function name), with `loc`(metrics) == `lines`(readability) required as a + sanity check; an ambiguous match (same path+name, disagreeing loc, e.g. two overloads) is dropped and + counted, not guessed at. + + 2. EXPOSURE, per function, read at the cutoff: `z` (the exact pre-sigmoid Posnett score, recomputed from + integer `toks=`/`vocab=` exactly as bench/readability_refactor_pairs.py does — lower z = less readable) + and `ccx` (cognitive complexity from `--metrics`, the same metric `src/quality.h`'s "complexity" gate + kind reads — higher ccx = more complex). + + 3. OUTCOME, defined before looking, over the window (CUTOFF, UNTIL]: + - "fix-shaped" commit: subject line (not body — §3c already found body-matching noisy) matches, case + insensitively, `\bfix(e[sd])?\b|\bbug(s|fix(e[sd])?)?\b|\bcrash(e[sd])?\b|\bregression(s)?\b`. + - a function counts as FIXED if any fix-shaped commit in the window has a diff hunk (old-file line + numbers, i.e. against that commit's OWN parent, not the cutoff) that overlaps the function's span + AS MEASURED AT THAT PARENT REVISION (re-scored per commit, so line drift from earlier window + commits cannot misattribute a hunk) — matched back to the population by (basename, function name). + A pure-insertion hunk (old count 0) is treated as touching whichever function(s) contain old-line + `start` or `start+1` (the two lines the insertion sits between). + - a function counts as MODIFIED (the broader arm) under the identical rule, over ALL commits that + touch a src file in the window, fix-shaped or not. Every fix-shaped commit is also a modifying + commit, so both arms are computed in one pass over the window's commits. + + 4. STATISTIC: fix-rate (and modified-rate) in the least-readable quartile (bottom 25% by z) vs the rest, + Wilson-interval per proportion, risk ratio with a Katz log-CI — reported beside the identical statistic + for the highest-ccx quartile vs the rest, on the SAME population, as the trusted-signal baseline. + Repeated within three size (lines-at-cutoff) terciles, with the least-readable/highest-ccx quartile + recomputed WITHIN each tercile, to test whether either signal survives controlling for size. + + 5. DECISION BANDS (restated from the paired doc's own pre-registration): the least-readable quartile's + raw fix-rate CI excludes a risk ratio of 1 and the effect does not vanish once stratified by size -> + keep the ordering claim as a weak actionable signal; effect present raw but gone within every size + tercile -> the size confound explains it, disclose and narrow the claim; CI includes 1 raw -> withdraw + the "predicts later fixes" claim outright. The complexity arm is reported for comparison, not as a bar + the lens must clear. + +House rule this script exists under (bench/ANSWERQUALITY.md, bench/BENCHMARK.md): a measurement harness is a +LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say; it is not wired into +test/regression.sh. + +Usage: + bench/readability_fixrate_validity.py --bin build/ripwire + bench/readability_fixrate_validity.py --bin build/ripwire --out pop_outcomes.tsv --json + +Deterministic given a fixed git history (pinned CUTOFF/UNTIL refs) and a fixed binary: no randomness anywhere. +""" + +from __future__ import annotations + +import argparse +import json +import math +import re +import shutil +import subprocess +import sys +import tarfile +import tempfile +import xml.etree.ElementTree as ET +from dataclasses import dataclass, field +from pathlib import Path +from typing import Optional + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import readability_refactor_pairs as rp # noqa: E402 (reuses git(), z_score(), DEFAULT_BIN) + +# Chosen to leave a multi-week, thousand-plus-commit follow-up window to v0.6.2 (2286 commits / ~17 days at +# the time this was pinned) — far enough back that a fix-shaped commit in the window cannot be an artifact of +# the cutoff being too recent, close enough that the population is still recognizably today's codebase. +DEFAULT_CUTOFF = "4f5c310c767f6f291c1db2e997b081f35b9a675d" +DEFAULT_UNTIL = "v0.6.2" + +FIX_RE = re.compile(r"\bfix(e[sd])?\b|\bbug(s|fix(e[sd])?)?\b|\bcrash(e[sd])?\b|\bregression(s)?\b", re.IGNORECASE) +HUNK_RE = re.compile(r"^@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@") + + +@dataclass +class FnRow: + path: str # relative to the population root (src/...) + name: str + start_line: int + lines: int + z: float + ccx: int + fixed: bool = False + modified: bool = False + + @property + def basename(self) -> str: + return Path(self.path).name + + @property + def end_line(self) -> int: + return self.start_line + self.lines - 1 + + +def run_xml(binary: str, root: Path, *flags: str) -> ET.Element: + proc = subprocess.run([binary, str(root), *flags], capture_output=True, text=True, timeout=120) + if proc.returncode != 0: + raise RuntimeError(f"{binary} {' '.join(flags)} on {root} exited {proc.returncode}: {proc.stderr.strip()}") + out = proc.stdout + start = out.index("<") + # skip the leading XML comment (legend) to reach the real root element + while out[start:start + 4] == "", start) + 3 + start = out.index("<", start) + return ET.fromstring(out[start:]) + + +def crawl_population(binary: str, src_root: Path) -> dict[tuple[str, str], FnRow]: + """Real crawl of the extracted src/ tree at the cutoff: match --readability rows to --metrics rows by + (path, name), requiring loc==lines to accept the match. Returns {(path, name): FnRow}.""" + read_root = run_xml(binary, src_root, "--readability", "--limit=100000") + met_root = run_xml(binary, src_root, "--metrics", "--top-k=100000") + + readab: dict[tuple[str, str], list[tuple[int, int, float]]] = {} # (path,name) -> [(start,lines,z)] + for fn in read_root.findall("fn"): + p = fn.get("p") + path, _, lineno = p.rpartition(":") + toks, vocab = int(fn.get("toks")), int(fn.get("vocab")) + vol_exact = toks * math.log2(vocab) if vocab > 0 else 0.0 + z = rp.z_score(vol_exact, int(fn.get("lines")), float(fn.get("ent"))) + readab.setdefault((path, fn.get("n")), []).append((int(lineno), int(fn.get("lines")), z)) + + metrics: dict[tuple[str, str], list[int]] = {} # (path,name) -> [loc, loc, ...] (one per sc match) + metrics_ccx: dict[tuple[str, str, int], int] = {} + for f in met_root.findall("f"): + path = f.get("p") + for s in f.findall("s"): + if s.get("t") not in ("fn", "method"): + continue + loc = int(s.get("loc")) + key = (path, s.get("n")) + metrics.setdefault(key, []).append(loc) + metrics_ccx[(path, s.get("n"), loc)] = int(s.get("ccx")) + + pop: dict[tuple[str, str], FnRow] = {} + ambiguous = 0 + for key, entries in readab.items(): + locs = metrics.get(key) + if not locs: + continue + for start, lines, z in entries: + if lines not in locs: + continue + ccx = metrics_ccx.get((key[0], key[1], lines)) + if ccx is None: + continue + if key in pop: + ambiguous += 1 + continue + pop[key] = FnRow(path=key[0], name=key[1], start_line=start, lines=lines, z=z, ccx=ccx) + print(f"# population: {len(pop)} matched functions ({ambiguous} ambiguous matches dropped)", file=sys.stderr) + return pop + + +def window_commits(until: str, cutoff: str) -> list[str]: + raw = rp.git("log", "--format=%H", f"{cutoff}..{until}", "--", + "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc") + return [ln for ln in raw.splitlines() if ln.strip()] + + +def parse_hunks(diff_text: str) -> list[tuple[int, int]]: + """[(old_start, old_end_inclusive)] for each hunk; a pure insertion (old count 0) becomes a + zero-width marker (start, start+1) so it can match a function containing either boundary line.""" + out = [] + for line in diff_text.splitlines(): + m = HUNK_RE.match(line) + if not m: + continue + old_start = int(m.group(1)) + old_count = int(m.group(2)) if m.group(2) is not None else 1 + if old_count == 0: + out.append((max(old_start, 1), max(old_start, 1) + 1)) + else: + out.append((old_start, old_start + old_count - 1)) + return out + + +def touched_functions_at_parent(binary: str, scratch: Path, sha: str, path: str) -> list[tuple[str, int, int]]: + """[(name, start, end)] for every function in `path` as it existed at sha^ (the state the diff's old + line numbers are relative to). Runs --readability directly (not rp.score_file) because that helper + drops the start line, which this function needs.""" + text = rp.show_file(f"{sha}^", path) + if text is None: + return [] + scratch_dir = scratch + if scratch_dir.exists(): + shutil.rmtree(scratch_dir) + scratch_dir.mkdir(parents=True) + (scratch_dir / Path(path).name).write_text(text, encoding="utf-8", errors="surrogateescape") + root = run_xml(binary, scratch_dir, "--readability", "--limit=100000") + out = [] + for fn in root.findall("fn"): + p = fn.get("p") + _, _, lineno = p.rpartition(":") + start = int(lineno) + lines = int(fn.get("lines")) + out.append((fn.get("n"), start, start + lines - 1)) + return out + + +def mark_outcomes(binary: str, pop: dict[tuple[str, str], FnRow], commits: list[str], scratch: Path) -> dict: + """One pass over every window commit that touches a src file: for each touched file, resolve the + function spans at that commit's PARENT, intersect against the diff's old-line hunks, and mark any + matching population function (by basename+name) as modified / (if the commit is fix-shaped) fixed.""" + by_basename_name: dict[tuple[str, str], list[FnRow]] = {} + for row in pop.values(): + by_basename_name.setdefault((row.basename, row.name), []).append(row) + + n_fix_commits = 0 + n_touch_commits = 0 + for sha in commits: + subject = rp.git("log", "-1", "--format=%s", sha).strip() + is_fix = bool(FIX_RE.search(subject)) + files = rp.touched_source_files(sha) + if not files: + continue + n_touch_commits += 1 + if is_fix: + n_fix_commits += 1 + for path in files: + diff = rp.git("diff", "-U0", f"{sha}^", sha, "--", path) + hunks = parse_hunks(diff) + if not hunks: + continue + spans = touched_functions_at_parent(binary, scratch, sha, path) + if not spans: + continue + basename = Path(path).name + for name, start, end in spans: + key = (basename, name) + rows = by_basename_name.get(key) + if not rows: + continue + touched = any(not (end < hs or start > he) for hs, he in hunks) + if not touched: + continue + for row in rows: + row.modified = True + if is_fix: + row.fixed = True + return {"fix_commits": n_fix_commits, "touch_commits": n_touch_commits, "window_total": len(commits)} + + +# ---------- statistics (stdlib only) ---------- + +def wilson_interval(k: int, n: int, z: float = 1.959963985) -> tuple[float, float, float]: + if n == 0: + return (float("nan"),) * 3 + p = k / n + denom = 1 + z * z / n + center = (p + z * z / (2 * n)) / denom + half = (z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n))) / denom + return p, max(0.0, center - half), min(1.0, center + half) + + +def risk_ratio_ci(k1: int, n1: int, k0: int, n0: int, z: float = 1.959963985) -> tuple[float, float, float]: + """Katz log-method CI for the risk ratio (group-1 rate / group-0 rate). NaNs out on a zero cell — + disclosed as such by the caller, never silently substituted.""" + if n1 == 0 or n0 == 0 or k1 == 0 or k0 == 0: + return (float("nan"),) * 3 + p1, p0 = k1 / n1, k0 / n0 + rr = p1 / p0 + se = math.sqrt((1 - p1) / (k1) + (1 - p0) / (k0)) if k1 and k0 else float("nan") + lo = rr * math.exp(-z * se) + hi = rr * math.exp(z * se) + return rr, lo, hi + + +def rate_block(flags: list[bool]) -> tuple[int, int, float]: + k = sum(1 for f in flags if f) + n = len(flags) + return k, n, (k / n if n else float("nan")) + + +def quartile_split(rows: list[FnRow], key) -> tuple[list[FnRow], list[FnRow]]: + """Bottom-quartile-by-key vs the rest. `key(row)` ascending; bottom 25% = worst by the study's own + convention (least readable = lowest z; most complex = HIGHEST ccx, so callers pass a negated key).""" + if not rows: + return [], [] + ordered = sorted(rows, key=key) + cut = max(1, round(len(ordered) * 0.25)) + return ordered[:cut], ordered[cut:] + + +def report_arm(label: str, worst: list[FnRow], rest: list[FnRow], outcome: str) -> dict: + wk, wn, wr = rate_block([getattr(r, outcome) for r in worst]) + rk, rn, rr_ = rate_block([getattr(r, outcome) for r in rest]) + _, w_lo, w_hi = wilson_interval(wk, wn) + _, r_lo, r_hi = wilson_interval(rk, rn) + rr, rr_lo, rr_hi = risk_ratio_ci(wk, wn, rk, rn) + return { + "label": label, "outcome": outcome, + "worst_k": wk, "worst_n": wn, "worst_rate": wr, "worst_ci": (w_lo, w_hi), + "rest_k": rk, "rest_n": rn, "rest_rate": rr_, "rest_ci": (r_lo, r_hi), + "risk_ratio": rr, "rr_ci": (rr_lo, rr_hi), + } + + +def size_terciles(rows: list[FnRow]) -> list[list[FnRow]]: + ordered = sorted(rows, key=lambda r: r.lines) + n = len(ordered) + t1 = ordered[: n // 3] + t2 = ordered[n // 3: 2 * n // 3] + t3 = ordered[2 * n // 3:] + return [t1, t2, t3] + + +def fmt_ci(ci: tuple[float, float]) -> str: + lo, hi = ci + if lo != lo or hi != hi: + return "n/a" + return f"[{lo:.3f}, {hi:.3f}]" + + +def print_report(pop: list[FnRow], meta: dict) -> None: + print(f"population: {len(pop)} functions at cutoff; window: {meta['window_total']} commits touching src, " + f"{meta['touch_commits']} with a matched src diff, {meta['fix_commits']} fix-shaped") + for outcome in ("fixed", "modified"): + print(f"\n=== outcome: {outcome} ===") + worst_z, rest_z = quartile_split(pop, key=lambda r: r.z) + worst_ccx, rest_ccx = quartile_split(pop, key=lambda r: -r.ccx) + for arm in (report_arm("readability (least-readable quartile)", worst_z, rest_z, outcome), + report_arm("complexity (highest-ccx quartile)", worst_ccx, rest_ccx, outcome)): + print(f" {arm['label']}: worst {arm['worst_k']}/{arm['worst_n']} = {arm['worst_rate']:.3f} " + f"{fmt_ci(arm['worst_ci'])} vs rest {arm['rest_k']}/{arm['rest_n']} = {arm['rest_rate']:.3f} " + f"{fmt_ci(arm['rest_ci'])} RR={arm['risk_ratio']:.3f} {fmt_ci(arm['rr_ci'])}") + print(" -- size-stratified (terciles by lines-at-cutoff, quartile recomputed within each) --") + for i, band in enumerate(size_terciles(pop), start=1): + wz, rz = quartile_split(band, key=lambda r: r.z) + wc, rc = quartile_split(band, key=lambda r: -r.ccx) + lo_lines = min(r.lines for r in band) if band else 0 + hi_lines = max(r.lines for r in band) if band else 0 + a = report_arm("readability", wz, rz, outcome) + b = report_arm("complexity", wc, rc, outcome) + print(f" T{i} (n={len(band)}, lines {lo_lines}-{hi_lines}):" + f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])} |" + f" complexity RR={b['risk_ratio']:.3f} {fmt_ci(b['rr_ci'])}") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=rp.DEFAULT_BIN) + ap.add_argument("--cutoff", default=DEFAULT_CUTOFF) + ap.add_argument("--until", default=DEFAULT_UNTIL) + ap.add_argument("--out", default=None, help="write the per-function TSV here") + ap.add_argument("--json", action="store_true") + ap.add_argument("--scratch", default=None) + args = ap.parse_args() + + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin}", file=sys.stderr) + return 2 + + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_fixrate_")) + try: + pop_src = scratch / "population_src" + pop_src.mkdir(parents=True) + archive = subprocess.run(["git", "archive", args.cutoff, "--", "src"], cwd=str(rp.ROOT), + capture_output=True, timeout=60) + if archive.returncode != 0: + print(f"error: git archive {args.cutoff} -- src failed: {archive.stderr.decode()}", file=sys.stderr) + return 2 + with tempfile.NamedTemporaryFile(suffix=".tar") as tf: + tf.write(archive.stdout) + tf.flush() + with tarfile.open(tf.name) as tar: + tar.extractall(pop_src) + + pop_map = crawl_population(args.bin, pop_src) + commits = window_commits(args.until, args.cutoff) + meta = mark_outcomes(args.bin, pop_map, commits, scratch / "mark_scratch") + meta["window_total"] = len(commits) + pop = list(pop_map.values()) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) + + if args.out: + with open(args.out, "w", encoding="utf-8") as f: + f.write("path\tname\tstart_line\tlines\tz\tccx\tfixed\tmodified\n") + for r in pop: + f.write(f"{r.path}\t{r.name}\t{r.start_line}\t{r.lines}\t{r.z:.6f}\t{r.ccx}\t{int(r.fixed)}\t{int(r.modified)}\n") + print(f"# wrote {len(pop)} rows to {args.out}", file=sys.stderr) + + if args.json: + def arm_json(rows, key): + worst, rest = quartile_split(rows, key=key) + return {o: report_arm("x", worst, rest, o) for o in ("fixed", "modified")} + print(json.dumps({ + "meta": meta, + "n_population": len(pop), + "readability": arm_json(pop, lambda r: r.z), + "complexity": arm_json(pop, lambda r: -r.ccx), + }, indent=2, default=str)) + else: + print_report(pop, meta) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index 0a1610446..41030450f 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -145,9 +145,9 @@ actually makes"): - **No better than chance, or a sign flip on a held-out language/corpus** → withdraw the ranking claim from any joined surface (`--ensemble`'s `rrank=`) and keep `--readability` only as a standalone, clearly-labeled "how does this formula see the codebase" report — the same demotion path `naminglens.h` already used once - (§4). + (§5). -We have not run the human-rated arm. It needs raters we do not have in this pass; §5 names it as the thing we +We have not run the human-rated arm. It needs raters we do not have in this pass; §6 names it as the thing we would like the most external help with. ## 3. What we ran WITHOUT human labels @@ -435,7 +435,80 @@ default) and `bench/readability_declared_pairs.py --bin build/ripwire --decompos the same pinned population as §3a itself). `--ref` points either command at a different immutable ref; `--until` still narrows within whichever ref is walked. -## 4. Precedent: we have already withdrawn a lens that failed exactly this kind of check +## 4. Does the ordering predict anything worth acting on? A later-fix-rate validation + +§2 and §3 both validate the ORDER: does it move the direction a refactor implies, is it stable under a +meaning-preserving rewrite. Neither asks the question an agent actually relies on when it reaches for +`--readability` to pick a target: **does a low score predict that a function will need fixing later?** That +is the implicit claim behind "use the worst-ranked functions as a worklist," and it has never been tested. +This section tests it, with no human label and no LLM judge, against a complexity control this repo's own +`--quality-delta` "complexity" gate kind already trusts, on the same population, with and without controlling +for function size. + +### Protocol (fixed in this commit, before the harness that computes it was run) + +This subsection was committed before `bench/readability_fixrate_validity.py` was run for the record — the +same discipline §3c used for proxy (c), and for the same reason: a band decided after seeing the number is +not a band. + +**Population.** Every `fn`/`method` in `src/**/*.{h,hpp,cpp,cc}` at a single pinned CUTOFF commit on +`v0.6.2`'s history — never `--all` (§3c's instrument-fix note explains why a shared `.git` makes that +non-reproducible). The cutoff is chosen to leave a multi-week, thousand-plus-commit follow-up window to the +pinned UNTIL ref (`v0.6.2` itself), long enough that a fix-shaped commit has real room to happen, short +enough that the population is still recognizably today's codebase. Functions are matched between a real +`--readability` crawl and a real `--metrics` crawl of the identical extracted tree (`git archive`, not +per-file scratch, so `--metrics`'s in/out/ccx see real cross-file structure) by (path, name), requiring +`loc`(metrics) `==` `lines`(readability); an ambiguous match (same path+name, disagreeing loc — almost always +an overload) is dropped and counted, never guessed at. + +**Exposure**, per function, read at the cutoff: `z`, the exact pre-sigmoid Posnett score recomputed from +integer `toks=`/`vocab=` exactly as `bench/readability_refactor_pairs.py` does (lower z = less readable — +`--readability`'s own least-readable-first sort order), and `ccx`, cognitive complexity from `--metrics` — +the same metric `src/quality.h`'s `"complexity"` gate kind reads, used here as the already-trusted baseline +on the identical population, not as a bar the lens must clear. + +**Outcome**, defined before looking, over the window `(CUTOFF, UNTIL]`: + +- **fix-shaped commit**: subject line (not body — §3c already found body-matching noisy) matches, case + insensitively: `\bfix(e[sd])?\b|\bbug(s|fix(e[sd])?)?\b|\bcrash(e[sd])?\b|\bregression(s)?\b`. +- a function is **FIXED** if any fix-shaped commit in the window has a diff hunk — in that commit's OWN + parent's line numbers, not the cutoff's — overlapping the function's span AS MEASURED AT THAT PARENT + revision (re-scored per commit specifically so line drift from earlier window commits cannot misattribute + a hunk to the wrong function), matched back to the population by (basename, function name). A pure-insertion + hunk (old count 0) is treated as touching whichever function(s) contain old-line `start` or `start+1`. +- a function is **MODIFIED** (the broader arm) under the identical rule over every commit that touches a src + file in the window, fix-shaped or not — a superset computed in the same pass, since every fix-shaped commit + is also a modifying commit. + +**Statistic.** Fix-rate (and modified-rate) in the least-readable quartile (bottom 25% by z) vs the rest, +Wilson interval per proportion, risk ratio with a Katz log-CI — reported beside the identical statistic for +the highest-ccx quartile vs the rest, on the same population, as the trusted-signal comparison point. +Repeated within three size (lines-at-cutoff) terciles, with the least-readable/highest-ccx quartile +recomputed WITHIN each tercile, to test whether either signal survives controlling for size. + +**Decision bands**, restated from this document's own §2 framing, applied to the raw (unstratified) arm: +the least-readable quartile's fix-rate CI excludes a risk ratio of 1, and the effect does not vanish once +stratified by size → keep the ordering claim as an actionable, if weak-to-moderate, signal; effect present +raw but gone in every size tercile → the size confound explains it, narrow the claim accordingly; CI includes +1 raw → withdraw the "predicts later fixes" claim outright. The complexity arm is reported for comparison, +never as a bar the lens must clear — Scalabrino/Trockman's own consensus (§2) is that no classic metric +correlates strongly with anything, so "about as good as complexity" is itself the "keep as a weak signal" +outcome, not a pass/fail line of its own. + +**A pre-registered caveat about the size control itself.** §3c already found that `z`'s direction on this +repository's own history is almost entirely the sign of the TOKEN-count change (96.0% of pairs), not the +line-count change — the Posnett lines coefficient is positive, so more lines alone would push z the other +way. Stratifying by LINES (the natural, legible size band) therefore does not fully neutralize the mechanism +§3c already implicated: two functions in the same line-count tercile can still differ sharply in token +volume, and z tracks that. A result that survives a lines-based stratification is not automatically a result +that survives a tokens-based one — that is disclosed here, before either number exists, as a limitation of +this design, not folded quietly into the verdict after the fact. + +Re-run: `bench/readability_fixrate_validity.py --bin build/ripwire` (population + both outcome arms, raw and +size-stratified, `v0.6.2`-pinned CUTOFF/UNTIL by default; `--out` writes the per-function TSV this section's +numbers were computed from). + +## 5. Precedent: we have already withdrawn a lens that failed exactly this kind of check This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of `docs/LINEAGE.md`, and the top of `src/naminglens.h` itself, record `naming-body-mismatch` — a rule that @@ -458,7 +531,7 @@ citing why this round's finding no longer applies. §3's numbers are not at that 63%-wrong-direction result on refactor commits is concerning enough that we think the human-rated study in §2 is now the right next step, not an optional nice-to-have. -## 5. What we would like help with +## 6. What we would like help with We are not readability researchers; we are reporting what a deterministic, disclosed, ordering-only lens measures against a construct it was never claimed to solve, and we would like informed pushback on the @@ -514,5 +587,5 @@ following, specifically: alongside the prompt-constraint study in the readability design note §7 as the joint reason `--quality-delta`'s gate is deterministic and external to the model, applied to the diff. - `docs/LINEAGE.md` §9.0 and `src/naminglens.h` (top-of-file comment) — the withdrawn `naming-body-mismatch` - rule, this repo's only other instance of "measured, then withdrawn," and the template §4 of this document + rule, this repo's only other instance of "measured, then withdrawn," and the template §5 of this document follows. From 8f93bbc38e30c27332c030c46ff970eb5f219858 Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 16:47:01 -0400 Subject: [PATCH 08/10] =?UTF-8?q?research(readability):=20later-fix-rate?= =?UTF-8?q?=20validation=20=E2=80=94=20results?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit bench/readability_fixrate_validity.py run against v0.6.2 history: 2,868 functions at cutoff 4f5c310c (2026-09-03), a 1,138-commit / 556-fix-shaped follow-up window to v0.6.2 (2026-09-21). Raw: least-readable-quartile risk ratio 2.87 [2.57, 3.21] for later fixes, 2.25 [2.08, 2.44] for any later modification — both larger than the complexity control's 2.26 / 1.93 on the identical population. The pre-registered "most likely outcome" (signal vanishes once size is held constant) did not happen: the readability RR clears 1 in five of six lines-stratified rows, only touching it in one (FIXED/T2, lower bound 0.995). Complexity inverts (RR significantly below 1) in the same T2 band on both outcomes, unanticipated by this protocol and reported as measured. Verdict per the pre-registered bands: keep the ordering claim as an actionable signal on this corpus, not withdraw it — with the disclosed caveat that a lines-based stratification does not fully rule out the token-volume mechanism §3c already found, so a token-count stratification is named as the next test, not run here. No src/ change. Co-Authored-By: Claude Sonnet 5 --- .../readability-construct-validity.md | 54 +++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index 41030450f..2c49e048e 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -508,6 +508,60 @@ Re-run: `bench/readability_fixrate_validity.py --bin build/ripwire` (population size-stratified, `v0.6.2`-pinned CUTOFF/UNTIL by default; `--out` writes the per-function TSV this section's numbers were computed from). +### Results + +**Population and window.** 2,868 of 3,004 functions crawled at the cutoff (`4f5c310c`, 2026-09-03) matched +between the `--readability` and `--metrics` passes (95.5%; 67 ambiguous matches dropped). The follow-up +window to `v0.6.2` (2026-09-21) held 1,138 commits touching a `src/` file, 556 of them fix-shaped by the +pre-registered regex (48.9%). + +**Raw (unstratified).** + +| outcome | arm | worst-quartile rate | rest rate | risk ratio (95% CI) | +|---|---|---|---|---| +| FIXED | readability (least-readable quartile, n=717) | 386/717 = 0.538 [0.502, 0.575] | 403/2151 = 0.187 [0.171, 0.204] | **2.87 [2.57, 3.21]** | +| FIXED | complexity (highest-ccx quartile, n=717) | 339/717 = 0.473 [0.437, 0.509] | 450/2151 = 0.209 [0.193, 0.227] | **2.26 [2.02, 2.53]** | +| MODIFIED | readability | 495/717 = 0.690 [0.656, 0.723] | 659/2151 = 0.306 [0.287, 0.326] | **2.25 [2.08, 2.44]** | +| MODIFIED | complexity | 452/717 = 0.630 [0.594, 0.665] | 702/2151 = 0.326 [0.307, 0.346] | **1.93 [1.78, 2.10]** | + +Both signals separate from 1 by a wide margin on this population, on both outcome definitions — and, raw, the +readability quartile's risk ratio is *larger* than the complexity quartile's on every row, not smaller. + +**Size-stratified** (terciles by lines-at-cutoff; n=956 each; quartile recomputed within each tercile): + +| tercile (lines) | outcome | readability RR (95% CI) | complexity RR (95% CI) | +|---|---|---|---| +| T1 (1–10) | FIXED | 3.30 [2.32, 4.70] | 0.80 [0.51, 1.24] | +| T2 (10–26) | FIXED | 1.27 [0.995, 1.63] | **0.55 [0.40, 0.77]** | +| T3 (27–1494) | FIXED | 1.90 [1.69, 2.14] | 1.39 [1.22, 1.58] | +| T1 (1–10) | MODIFIED | 2.39 [1.88, 3.05] | 1.08 [0.82, 1.44] | +| T2 (10–26) | MODIFIED | 1.23 [1.03, 1.47] | **0.79 [0.64, 0.98]** | +| T3 (27–1494) | MODIFIED | 1.56 [1.44, 1.69] | 1.24 [1.13, 1.37] | + +**Read this plainly.** The pre-registered "most likely outcome" — the signal vanishing once size is held +constant — did **not** happen. The readability risk ratio stays above 1, with a CI excluding 1, in five of +six stratified rows; it only touches 1 in the FIXED/T2 row (lower bound 0.995), and even there the MODIFIED/T2 +row for the same tercile clears 1 (1.03). It attenuates from the T1 (smallest-function) band — where it is +strongest, 3.30 and 2.39 — through T2, then rises again in T3. Complexity's within-band behavior is the more +surprising result here: it is flat-to-inverted in T1 and *significantly below 1* in T2 (0.55 and 0.79, both +CIs excluding 1) — meaning, within these two size bands, the highest-cognitive-complexity quartile was +fixed/modified *less* often than the rest, the opposite of the trusted baseline's raw-population direction. +Complexity only behaves as expected (RR > 1) in T3, the largest-function band. + +**What this does and does not show.** On the raw population, `--readability`'s ordering separates a fixed +population by later-fix rate at least as well as the complexity control does, and the effect survives a +lines-based size stratification better than complexity's own effect does. That is a real, actionable-looking +signal on this corpus, under this outcome definition — the decision band above calls this a "keep," not a +"withdraw." But the pre-registered caveat means this is not the full size-confound test: `z` is dominated by +token volume, not lines (§3c), so a function can sit in the smallest LINE tercile while carrying a large +TOKEN volume relative to its tercile-mates, and the lines-based stratification cannot separate "z predicts +fixes independent of size" from "z still partly tracks a size axis lines does not capture." A token-count +stratification (bucketing by `toks=` instead of `lines=`) is the natural next test and was not run in this +round. Separately: complexity's inversion in T2 is itself worth a second look before this repository leans on +"highest complexity" as a worklist filter for medium-sized functions — that result was not anticipated by +this protocol and is reported exactly as measured, not smoothed over because it complicates the expected +story. + ## 5. Precedent: we have already withdrawn a lens that failed exactly this kind of check This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of From 36fc872bdfbe27c0ae1aa4570c0cecf00a25aea7 Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 17:14:13 -0400 Subject: [PATCH 09/10] research(readability): correct the fix-rate stratification to token-count MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The line-based size stratification in the prior commit did not hold the confound it set out to test: §3c already found z tracks the SIGN of the token-count change on 96.0% of this repo's own refactor pairs, not the line-count change, so a lines tercile does not hold z's own size axis constant. Adds a toks= (Halstead N, --readability's own count, the exact integer the lens's volume term is computed from) stratification beside the existing lines one on the identical population/outcomes. The two disagree, as intended: line-stratified, readability's RR excluded 1 in 5/6 rows; token-stratified (the decisive one), it excludes 1 in only 4/6 - both T2 (middle-third-by-tokens) rows now include 1 (1.06 [0.82,1.36] fixed, 1.05 [0.88,1.26] modified), where the line version had reported a borderline signal. Verdict restated against the pre-registered bands: not "keep exactly as-is" (requires no vanishing) and not "withdraw" (T1 and T3 both still clear RR>1 with CIs excluding 1) - this document's own §2 middle band, "keep but narrow": the later-fix signal is real at the size extremes and silent in the middle third, so the raw-population RR overstates what an agent should expect from a mid-sized function. Complexity's T2 inversion, checked for a file or symbol-kind concentration (max file share 5%, max keyword share 3.3% across ~40 files in both the line- and token-based T2 highest-ccx quartiles): no cheap explanation found, left as measured. Adds an external-validity line: every number in this section is this repository's own AI-authored, heavily-gated history, not a general claim about code. No src/ change; the pre-registration and first-results commits are untouched. Co-Authored-By: Claude Sonnet 5 --- bench/readability_fixrate_validity.py | 66 ++++++++++----- .../readability-construct-validity.md | 81 +++++++++++++++++++ 2 files changed, 126 insertions(+), 21 deletions(-) diff --git a/bench/readability_fixrate_validity.py b/bench/readability_fixrate_validity.py index 7b2ec4c28..73dd5de68 100644 --- a/bench/readability_fixrate_validity.py +++ b/bench/readability_fixrate_validity.py @@ -42,8 +42,14 @@ 4. STATISTIC: fix-rate (and modified-rate) in the least-readable quartile (bottom 25% by z) vs the rest, Wilson-interval per proportion, risk ratio with a Katz log-CI — reported beside the identical statistic for the highest-ccx quartile vs the rest, on the SAME population, as the trusted-signal baseline. - Repeated within three size (lines-at-cutoff) terciles, with the least-readable/highest-ccx quartile - recomputed WITHIN each tercile, to test whether either signal survives controlling for size. + Repeated within three size terciles TWICE: once by `lines`-at-cutoff, once by `toks`-at-cutoff (Halstead + N, --readability's own toks= — the size measure the lens actually consumes for its volume term, not a + proxy for it), with the least-readable/highest-ccx quartile recomputed WITHIN each tercile. The first + round of this script (see the paired doc's §4) stratified by lines only; a lines-based stratification + does not hold the confound constant, because z tracks the SIGN of the token-count change on 96.0% of + this repo's own refactor pairs (§3c), not the line-count change — a function can gain tokens with no + line change at all. The token-stratified arm is the decisive one for the size-confound question; the + line-stratified arm is kept alongside it because the two disagreeing would itself be informative. 5. DECISION BANDS (restated from the paired doc's own pre-registration): the least-readable quartile's raw fix-rate CI excludes a risk ratio of 1 and the effect does not vanish once stratified by size -> @@ -98,6 +104,9 @@ class FnRow: name: str start_line: int lines: int + toks: int # Halstead N (operator+operand count) — --readability's own toks=, the exact + # integer z's volume term is computed from; the size measure the lens actually + # consumes, not a proxy for it. z: float ccx: int fixed: bool = False @@ -131,14 +140,14 @@ def crawl_population(binary: str, src_root: Path) -> dict[tuple[str, str], FnRow read_root = run_xml(binary, src_root, "--readability", "--limit=100000") met_root = run_xml(binary, src_root, "--metrics", "--top-k=100000") - readab: dict[tuple[str, str], list[tuple[int, int, float]]] = {} # (path,name) -> [(start,lines,z)] + readab: dict[tuple[str, str], list[tuple[int, int, int, float]]] = {} # (path,name) -> [(start,lines,toks,z)] for fn in read_root.findall("fn"): p = fn.get("p") path, _, lineno = p.rpartition(":") toks, vocab = int(fn.get("toks")), int(fn.get("vocab")) vol_exact = toks * math.log2(vocab) if vocab > 0 else 0.0 z = rp.z_score(vol_exact, int(fn.get("lines")), float(fn.get("ent"))) - readab.setdefault((path, fn.get("n")), []).append((int(lineno), int(fn.get("lines")), z)) + readab.setdefault((path, fn.get("n")), []).append((int(lineno), int(fn.get("lines")), toks, z)) metrics: dict[tuple[str, str], list[int]] = {} # (path,name) -> [loc, loc, ...] (one per sc match) metrics_ccx: dict[tuple[str, str, int], int] = {} @@ -158,7 +167,7 @@ def crawl_population(binary: str, src_root: Path) -> dict[tuple[str, str], FnRow locs = metrics.get(key) if not locs: continue - for start, lines, z in entries: + for start, lines, toks, z in entries: if lines not in locs: continue ccx = metrics_ccx.get((key[0], key[1], lines)) @@ -167,7 +176,7 @@ def crawl_population(binary: str, src_root: Path) -> dict[tuple[str, str], FnRow if key in pop: ambiguous += 1 continue - pop[key] = FnRow(path=key[0], name=key[1], start_line=start, lines=lines, z=z, ccx=ccx) + pop[key] = FnRow(path=key[0], name=key[1], start_line=start, lines=lines, toks=toks, z=z, ccx=ccx) print(f"# population: {len(pop)} matched functions ({ambiguous} ambiguous matches dropped)", file=sys.stderr) return pop @@ -316,8 +325,10 @@ def report_arm(label: str, worst: list[FnRow], rest: list[FnRow], outcome: str) } -def size_terciles(rows: list[FnRow]) -> list[list[FnRow]]: - ordered = sorted(rows, key=lambda r: r.lines) +def size_terciles(rows: list[FnRow], key=lambda r: r.lines) -> list[list[FnRow]]: + """Terciles by `key` (default: lines-at-cutoff; pass `lambda r: r.toks` for the token-count measure + the lens itself consumes).""" + ordered = sorted(rows, key=key) n = len(ordered) t1 = ordered[: n // 3] t2 = ordered[n // 3: 2 * n // 3] @@ -344,17 +355,21 @@ def print_report(pop: list[FnRow], meta: dict) -> None: print(f" {arm['label']}: worst {arm['worst_k']}/{arm['worst_n']} = {arm['worst_rate']:.3f} " f"{fmt_ci(arm['worst_ci'])} vs rest {arm['rest_k']}/{arm['rest_n']} = {arm['rest_rate']:.3f} " f"{fmt_ci(arm['rest_ci'])} RR={arm['risk_ratio']:.3f} {fmt_ci(arm['rr_ci'])}") - print(" -- size-stratified (terciles by lines-at-cutoff, quartile recomputed within each) --") - for i, band in enumerate(size_terciles(pop), start=1): - wz, rz = quartile_split(band, key=lambda r: r.z) - wc, rc = quartile_split(band, key=lambda r: -r.ccx) - lo_lines = min(r.lines for r in band) if band else 0 - hi_lines = max(r.lines for r in band) if band else 0 - a = report_arm("readability", wz, rz, outcome) - b = report_arm("complexity", wc, rc, outcome) - print(f" T{i} (n={len(band)}, lines {lo_lines}-{hi_lines}):" - f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])} |" - f" complexity RR={b['risk_ratio']:.3f} {fmt_ci(b['rr_ci'])}") + for strat_label, strat_key, unit_key, unit_name in ( + ("lines-at-cutoff", (lambda r: r.lines), (lambda r: r.lines), "lines"), + ("toks-at-cutoff (Halstead N, --readability's own toks=)", (lambda r: r.toks), (lambda r: r.toks), "toks"), + ): + print(f" -- size-stratified (terciles by {strat_label}, quartile recomputed within each) --") + for i, band in enumerate(size_terciles(pop, key=strat_key), start=1): + wz, rz = quartile_split(band, key=lambda r: r.z) + wc, rc = quartile_split(band, key=lambda r: -r.ccx) + lo_u = min(unit_key(r) for r in band) if band else 0 + hi_u = max(unit_key(r) for r in band) if band else 0 + a = report_arm("readability", wz, rz, outcome) + b = report_arm("complexity", wc, rc, outcome) + print(f" T{i} (n={len(band)}, {unit_name} {lo_u}-{hi_u}):" + f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])} |" + f" complexity RR={b['risk_ratio']:.3f} {fmt_ci(b['rr_ci'])}") def main() -> int: @@ -397,20 +412,29 @@ def main() -> int: if args.out: with open(args.out, "w", encoding="utf-8") as f: - f.write("path\tname\tstart_line\tlines\tz\tccx\tfixed\tmodified\n") + f.write("path\tname\tstart_line\tlines\ttoks\tz\tccx\tfixed\tmodified\n") for r in pop: - f.write(f"{r.path}\t{r.name}\t{r.start_line}\t{r.lines}\t{r.z:.6f}\t{r.ccx}\t{int(r.fixed)}\t{int(r.modified)}\n") + f.write(f"{r.path}\t{r.name}\t{r.start_line}\t{r.lines}\t{r.toks}\t{r.z:.6f}\t{r.ccx}\t{int(r.fixed)}\t{int(r.modified)}\n") print(f"# wrote {len(pop)} rows to {args.out}", file=sys.stderr) if args.json: def arm_json(rows, key): worst, rest = quartile_split(rows, key=key) return {o: report_arm("x", worst, rest, o) for o in ("fixed", "modified")} + def strat_json(strat_key): + return [ + {"n": len(band), + "readability": arm_json(band, lambda r: r.z), + "complexity": arm_json(band, lambda r: -r.ccx)} + for band in size_terciles(pop, key=strat_key) + ] print(json.dumps({ "meta": meta, "n_population": len(pop), "readability": arm_json(pop, lambda r: r.z), "complexity": arm_json(pop, lambda r: -r.ccx), + "stratified_by_lines": strat_json(lambda r: r.lines), + "stratified_by_toks": strat_json(lambda r: r.toks), }, indent=2, default=str)) else: print_report(pop, meta) diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index 2c49e048e..a9088df64 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -562,6 +562,87 @@ round. Separately: complexity's inversion in T2 is itself worth a second look be this protocol and is reported exactly as measured, not smoothed over because it complicates the expected story. +### Correction: the line-based stratification did not hold the confound constant + +The paragraph above named the gap and deferred the fix; this section runs it. The lines-based stratification +just reported does not hold the actual confound constant: §3c already established that `z`'s direction +tracks the SIGN of the token-count change on 96.0% of this repository's own refactor pairs, not the +line-count change — the median wrong-direction pair there was +5 tokens, 0 lines. A function can gain a +meaningful share of a size band's token volume with no line-count change at all, so two functions in the +same LINE tercile can differ sharply in the unit `z` actually consumes. "Keep the ordering claim" was +therefore not yet earned by the line-stratified result alone. This re-runs the identical stratified analysis +by `toks=` — the Halstead N (operator+operand count) `--readability` itself emits, the exact integer the +lens's own volume term is computed from, not a re-derived proxy — on the same population, same outcome +definitions, same CIs, using the updated `bench/readability_fixrate_validity.py` (now captures `toks=` per +function and stratifies by either measure). + +**Token-count stratification** (terciles by `toks`-at-cutoff; n=956 each; quartile recomputed within each +tercile), reported beside the line-stratified table rather than replacing it: + +| tercile (toks) | outcome | readability RR (95% CI) | complexity RR (95% CI) | +|---|---|---|---| +| T1 (10–72) | FIXED | 2.28 [1.56, 3.33] | 1.01 [0.65, 1.57] | +| T2 (73–191) | FIXED | 1.06 [0.82, 1.36] | **0.62 [0.46, 0.84]** | +| T3 (191–8023) | FIXED | 1.89 [1.68, 2.13] | 1.38 [1.21, 1.58] | +| T1 (10–72) | MODIFIED | 1.74 [1.33, 2.28] | 1.02 [0.75, 1.39] | +| T2 (73–191) | MODIFIED | 1.05 [0.88, 1.26] | 0.85 [0.70, 1.04] | +| T3 (191–8023) | MODIFIED | 1.59 [1.46, 1.72] | 1.27 [1.15, 1.39] | + +**Do the two stratifications agree?** No, and the disagreement is itself the finding. Line-stratified, +readability's RR excluded 1 in five of six rows (only FIXED/T2 touched 1, at a lower bound of 0.995). +Token-stratified — the measure the lens actually consumes — readability's RR excludes 1 in only **four** of +six rows: it holds in T1 and T3 on both outcomes, but **both** T2 rows now include 1 (FIXED 1.06 [0.82, +1.36]; MODIFIED 1.05 [0.88, 1.26]) — indistinguishable from no effect in the middle third of the token +distribution, where the line-based version had reported a real (if borderline) signal. The line-based +analysis was, exactly as suspected, partly reading a token-volume effect that a line-count band does not +hold constant. + +**Restated verdict against the pre-registered bands, using the token-stratified result as decisive.** The raw +(unstratified) separation stands unchanged (RR 2.87 [2.57, 3.21] fixed, 2.25 [2.08, 2.44] modified — both +exceed the complexity control's 2.26 / 1.93 on the same population) and by itself would satisfy the "keep" +band. But the token-stratified arm shows that separation is not uniform across the size range the lens +itself measures: it is real and CI-excludes-1 at both ends (small-token and large-token thirds) and +statistically silent in the middle third. That is neither this document's "keep exactly as-is" band (which +requires the effect not to vanish under stratification, full stop) nor its "withdraw" band (which requires +it to vanish in *every* tercile — it does not: T1 and T3 both hold). It matches the **middle band this +document's own §2 already defined for exactly this shape of result**: "consistent on some function shapes +and not others… disclose that the lens is known to track [size] more than it tracks anything +[size]-independent, refined by what was measured." The verdict is therefore: **keep the ordering claim, but +narrowed** — `--readability`'s later-fix signal on this corpus is not a uniform property of the ranking, it +is concentrated at the size extremes (very small and very large functions, measured in tokens) and +disappears for functions of middling token volume, where the raw quartile RR reported earlier is optimistic. +An agent reading "least readable" as a worklist filter should expect the signal to be weakest for +run-of-the-mill mid-sized functions — which is most of the codebase (T2 by construction) — and strongest at +the tails. This is a materially weaker claim than "keep exactly as-is," and this document says so plainly +rather than defaulting to the friendlier reading. + +**T2 diagnosis: is the complexity inversion explained by a file or symbol-kind concentration?** Complexity's +below-1 risk ratio in the middle tercile is the more surprising result in both stratifications (line-based: +0.55 fixed / 0.79 modified; token-based: 0.62 fixed / 0.85 modified, the modified arm no longer excluding 1 +under the token version). Checked cheaply, on both the line-T2 and token-T2 highest-ccx quartiles (n=239 +each): no file supplies more than 5% of either group (top file `src/docdrift.h` at 12–15 of 239), and no +name-pattern keyword (`test`, `match`, `find`, `build`, `table`, `gen`, `classify`, `parse`, `check`, …) +covers more than 3.3% — nothing resembling generated code, a table literal, or a test-fixture cluster +dominates either quartile. The sampled names instead look like a broad mix of short, branch-dense helpers +(classifiers, matchers, small validators: `walkAggregateBody`, `classifySkipHealth`, +`sliceBuildAnchorOccs`) spread across ~40 different files with no concentration. **This is reported as +unexplained.** A plausible but unverified guess — that dense, short decision functions get written once, +exhaustively branch-tested, and rarely revisited — is not checked here and is not asserted as a finding; per +this document's own rule, an anomaly without a cheap explanation is left as measured, not smoothed into a +story. + +**External validity.** Every number in this section, both stratifications and the raw arm, comes from one +repository's own history: `ripwire`'s own `src/` under this project's ~662 CI gates, quality-delta review and +adversarial CI, much of it AI-authored under those gates. A later-fix rate measured here is a claim about +*this corpus under this development process*, not a general claim about code, or even about AI-authored code +generally — a codebase without this repository's gate density, review discipline, or authorship mix could see +a different relationship (or none) between `z` and later-fix rate. Nothing in this section should be read as +"readability predicts fixes" as a general software-engineering claim; it is "readability predicted fixes on +this codebase, in this window, measured this way" — exactly the same scope limit §3's proxies already carry. + +Re-run (updated to also emit the token-stratified table and the per-function `toks=` column in `--out`): +`bench/readability_fixrate_validity.py --bin build/ripwire`. + ## 5. Precedent: we have already withdrawn a lens that failed exactly this kind of check This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of From a1b1930adc307f0b684e4018879c7bf6913fe023 Mon Sep 17 00:00:00 2001 From: joyful-ii-V-I Date: Tue, 22 Sep 2026 17:21:51 -0400 Subject: [PATCH 10/10] =?UTF-8?q?research(readability):=20withdraw=20the?= =?UTF-8?q?=20later-fix=20claim=20=E2=80=94=20terciles=20don't=20hold=20si?= =?UTF-8?q?ze?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit POST-HOC (prompted by looking at the tercile table, not pre-registered): the token terciles that showed a signal are also the widest internally (T1 7.2x, T3 42x); the silent one (T2) is the narrowest (2.6x). A tercile does not hold the lens's own size unit constant, so "survives at both extremes" is what a residual size effect predicts too, not only what an independent readability effect would. Adds decile-by-toks stratification (10 bands instead of 3) plus internal range per band to bench/readability_fixrate_validity.py, and a --load-tsv flag to re-stratify already-computed data in under a second instead of repeating the ~15-minute crawl. Result: 8 of 10 deciles (the genuinely narrow ones, 1.3x-2.9x internal range) show no effect on either outcome - CIs include 1, point estimates scattered both sides of 1. The two deciles that remain significant are the two with residual size range left inside them (D10 unambiguously at 16.3x; D09 flagged as not cleanly resolved at n=287/decile). A nearest-token-neighbour matched-control check was also attempted and found degenerate (46 distinct matches for 717 quartile members, top 2 matches covering 73.8%) - reported as discarded, not as evidence. Verdict restated against the pre-registered bands using deciles as decisive: withdraw, not "keep, narrowed" as the prior commit concluded. The raw and tercile-level separation is better explained as a residual size effect in the lens's own units than an independent later-fix signal - the same token-count mechanism §3c already found driving the lens's refactor-pair direction. The prior "keep, narrowed" verdict is superseded in this document, not rewritten. Pre-registration and both earlier results commits are untouched. No src/ change. Co-Authored-By: Claude Sonnet 5 --- bench/readability_fixrate_validity.py | 141 +++++++++++++----- .../readability-construct-validity.md | 85 +++++++++++ 2 files changed, 192 insertions(+), 34 deletions(-) diff --git a/bench/readability_fixrate_validity.py b/bench/readability_fixrate_validity.py index 73dd5de68..cb4888c20 100644 --- a/bench/readability_fixrate_validity.py +++ b/bench/readability_fixrate_validity.py @@ -62,9 +62,19 @@ LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say; it is not wired into test/regression.sh. +POST-HOC ADDITION (labelled as such, not pre-registered — prompted after seeing the tercile-stratified +result disagree with the line-stratified one): terciles are wide enough that neither one holds the lens's +own size unit (toks) constant within a band, so the script also reports (a) DECILES by toks — ten narrower +bands, to see whether the effect shrinks toward RR=1 as the band narrows, which is what a residual-size +effect would do and a genuine per-token-count-controlled readability effect would not — beside each +decile's own internal token range, since "does RR track the range" is the actual question; and (b) a +nearest-token-neighbour matched control: each least-readable-quartile function against its closest-toks +match from the rest of the population, one-to-one, rather than against a whole band. + Usage: bench/readability_fixrate_validity.py --bin build/ripwire bench/readability_fixrate_validity.py --bin build/ripwire --out pop_outcomes.tsv --json + bench/readability_fixrate_validity.py --load-tsv pop_outcomes.tsv # re-stratify already-computed data Deterministic given a fixed git history (pinned CUTOFF/UNTIL refs) and a fixed binary: no randomness anywhere. """ @@ -72,6 +82,8 @@ from __future__ import annotations import argparse +import bisect +import csv import json import math import re @@ -325,15 +337,38 @@ def report_arm(label: str, worst: list[FnRow], rest: list[FnRow], outcome: str) } -def size_terciles(rows: list[FnRow], key=lambda r: r.lines) -> list[list[FnRow]]: - """Terciles by `key` (default: lines-at-cutoff; pass `lambda r: r.toks` for the token-count measure - the lens itself consumes).""" +def size_ntiles(rows: list[FnRow], key=lambda r: r.lines, n: int = 3) -> list[list[FnRow]]: + """N equal-count bands by `key` (default: lines-at-cutoff, 3 = terciles; pass `lambda r: r.toks` for + the token-count measure the lens itself consumes, and n=10 for deciles — a POST-HOC refinement, see + the paired doc's third §4 subsection: terciles are wide enough that a tercile does not hold size + constant on its own, so a real readability effect and a residual-size effect both predict the same + tercile-level RR pattern; only a narrower band distinguishes them).""" ordered = sorted(rows, key=key) - n = len(ordered) - t1 = ordered[: n // 3] - t2 = ordered[n // 3: 2 * n // 3] - t3 = ordered[2 * n // 3:] - return [t1, t2, t3] + total = len(ordered) + bounds = [round(i * total / n) for i in range(n + 1)] + return [ordered[bounds[i]:bounds[i + 1]] for i in range(n)] + + +def size_terciles(rows: list[FnRow], key=lambda r: r.lines) -> list[list[FnRow]]: + return size_ntiles(rows, key=key, n=3) + + +def nearest_token_match(pop: list[FnRow], quartile: list[FnRow]) -> list[FnRow]: + """For each function in `quartile`, its nearest neighbour by `toks` among `pop` functions NOT in the + quartile (ties broken by the earlier one in token-sorted order; matching is WITH replacement — a + popular token count can supply more than one match, disclosed at the call site). A cheap alternative + to a fixed-width band: instead of asking "does the effect survive in a band this wide," it asks "does + it survive against a control matched almost exactly on the lens's own size unit.\"""" + quartile_keys = {id(r) for r in quartile} + rest_sorted = sorted((r for r in pop if id(r) not in quartile_keys), key=lambda r: r.toks) + rest_toks = [r.toks for r in rest_sorted] + matches = [] + for q in quartile: + i = bisect.bisect_left(rest_toks, q.toks) + candidates = [j for j in (i - 1, i) if 0 <= j < len(rest_sorted)] + best = min(candidates, key=lambda j: abs(rest_sorted[j].toks - q.toks)) + matches.append(rest_sorted[best]) + return matches def fmt_ci(ci: tuple[float, float]) -> str: @@ -370,6 +405,29 @@ def print_report(pop: list[FnRow], meta: dict) -> None: print(f" T{i} (n={len(band)}, {unit_name} {lo_u}-{hi_u}):" f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])} |" f" complexity RR={b['risk_ratio']:.3f} {fmt_ci(b['rr_ci'])}") + print(" -- POST-HOC (prompted by the tercile result, not pre-registered): " + "DECILES by toks-at-cutoff, quartile recomputed within each decile --") + for i, band in enumerate(size_ntiles(pop, key=lambda r: r.toks, n=10), start=1): + wz, rz = quartile_split(band, key=lambda r: r.z) + a = report_arm("readability", wz, rz, outcome) + lo_t = min(r.toks for r in band) if band else 0 + hi_t = max(r.toks for r in band) if band else 0 + ratio = (hi_t / lo_t) if lo_t > 0 else float("inf") + print(f" D{i:02d} (n={len(band)}, toks {lo_t}-{hi_t}, internal range {ratio:.2f}x):" + f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])}") + print(" -- POST-HOC: nearest-token-neighbour matched control " + "(least-readable quartile vs. its 1-NN-by-toks match from the rest) --") + worst_z, _rest_z = quartile_split(pop, key=lambda r: r.z) + matched = nearest_token_match(pop, worst_z) + wk, wn, wr = rate_block([getattr(r, outcome) for r in worst_z]) + mk, mn, mr = rate_block([getattr(r, outcome) for r in matched]) + _, w_lo, w_hi = wilson_interval(wk, wn) + _, m_lo, m_hi = wilson_interval(mk, mn) + rr, rr_lo, rr_hi = risk_ratio_ci(wk, wn, mk, mn) + mean_gap = (sum(r.toks for r in worst_z) - sum(r.toks for r in matched)) / len(worst_z) if worst_z else 0.0 + print(f" least-readable quartile {wk}/{wn} = {wr:.3f} {fmt_ci((w_lo, w_hi))} vs matched control " + f"{mk}/{mn} = {mr:.3f} {fmt_ci((m_lo, m_hi))} RR={rr:.3f} {fmt_ci((rr_lo, rr_hi))}" + f" (mean token gap quartile-minus-match: {mean_gap:+.1f})") def main() -> int: @@ -380,35 +438,50 @@ def main() -> int: ap.add_argument("--out", default=None, help="write the per-function TSV here") ap.add_argument("--json", action="store_true") ap.add_argument("--scratch", default=None) + ap.add_argument("--load-tsv", default=None, + help="skip the population crawl and outcome-marking pass (the slow ~15-minute step) " + "and re-load a previously written --out TSV instead — for re-stratifying " + "(--strata, deciles, the matched control) on data already computed. The window/" + "fix-commit counts in the report line are not available from a TSV alone and " + "print as 'n/a'.") args = ap.parse_args() - if not Path(args.bin).is_file(): - print(f"error: ripwire binary not found at {args.bin}", file=sys.stderr) - return 2 - - scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_fixrate_")) - try: - pop_src = scratch / "population_src" - pop_src.mkdir(parents=True) - archive = subprocess.run(["git", "archive", args.cutoff, "--", "src"], cwd=str(rp.ROOT), - capture_output=True, timeout=60) - if archive.returncode != 0: - print(f"error: git archive {args.cutoff} -- src failed: {archive.stderr.decode()}", file=sys.stderr) + if args.load_tsv: + pop = [] + with open(args.load_tsv, encoding="utf-8") as f: + for row in csv.DictReader(f, delimiter="\t"): + pop.append(FnRow(path=row["path"], name=row["name"], start_line=int(row["start_line"]), + lines=int(row["lines"]), toks=int(row["toks"]), z=float(row["z"]), + ccx=int(row["ccx"]), fixed=bool(int(row["fixed"])), + modified=bool(int(row["modified"])))) + meta = {"window_total": "n/a (--load-tsv)", "touch_commits": "n/a", "fix_commits": "n/a"} + else: + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin}", file=sys.stderr) return 2 - with tempfile.NamedTemporaryFile(suffix=".tar") as tf: - tf.write(archive.stdout) - tf.flush() - with tarfile.open(tf.name) as tar: - tar.extractall(pop_src) - - pop_map = crawl_population(args.bin, pop_src) - commits = window_commits(args.until, args.cutoff) - meta = mark_outcomes(args.bin, pop_map, commits, scratch / "mark_scratch") - meta["window_total"] = len(commits) - pop = list(pop_map.values()) - finally: - if not args.scratch: - shutil.rmtree(scratch, ignore_errors=True) + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_fixrate_")) + try: + pop_src = scratch / "population_src" + pop_src.mkdir(parents=True) + archive = subprocess.run(["git", "archive", args.cutoff, "--", "src"], cwd=str(rp.ROOT), + capture_output=True, timeout=60) + if archive.returncode != 0: + print(f"error: git archive {args.cutoff} -- src failed: {archive.stderr.decode()}", file=sys.stderr) + return 2 + with tempfile.NamedTemporaryFile(suffix=".tar") as tf: + tf.write(archive.stdout) + tf.flush() + with tarfile.open(tf.name) as tar: + tar.extractall(pop_src) + + pop_map = crawl_population(args.bin, pop_src) + commits = window_commits(args.until, args.cutoff) + meta = mark_outcomes(args.bin, pop_map, commits, scratch / "mark_scratch") + meta["window_total"] = len(commits) + pop = list(pop_map.values()) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) if args.out: with open(args.out, "w", encoding="utf-8") as f: diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md index a9088df64..2b57a793d 100644 --- a/docs/research/readability-construct-validity.md +++ b/docs/research/readability-construct-validity.md @@ -643,6 +643,91 @@ this codebase, in this window, measured this way" — exactly the same scope lim Re-run (updated to also emit the token-stratified table and the per-function `toks=` column in `--out`): `bench/readability_fixrate_validity.py --bin build/ripwire`. +### Second correction (POST-HOC, not pre-registered): the tercile itself does not hold size constant + +**This subsection was prompted by looking at the token-stratified table above**, not written before it — +labelled as such, per this document's own rule about what counts as pre-registration. The tercile result +has a simpler reading than "keep, narrowed": T1 (10–72 tokens) spans a 7.2× internal range, T2 (73–191) spans +2.6×, T3 (191–8023) spans 42×. The two terciles that show a readability signal are the two with the widest +internal spread; the one that is silent is by far the narrowest. A tercile — even a token tercile — does not +hold size constant, so "the effect survives at both extremes" is exactly what a *residual* size effect +predicts too, not only what an independent readability effect would predict. The two hypotheses make +different, testable predictions on narrower bands: a genuine readability effect should not shrink as the +band narrows; a residual-size effect should shrink toward RR=1, at a rate that tracks how much internal +range is left in the band. + +**Deciles by `toks`-at-cutoff** (same population, same outcome definitions, same CIs; quartile recomputed +within each decile; `bench/readability_fixrate_validity.py --load-tsv pop_outcomes.tsv` re-derives this table +from the already-computed per-function data in under a second — no re-crawl needed): + +| decile | toks range | internal range | n | FIXED RR (95% CI) | MODIFIED RR (95% CI) | +|---|---|---|---|---|---| +| D01 | 10–29 | 2.90× | 287 | 1.12 [0.31, 4.11] | 0.66 [0.23, 1.90] | +| D02 | 29–47 | 1.62× | 287 | 1.49 [0.70, 3.18] | 1.27 [0.77, 2.09] | +| D03 | 47–66 | 1.40× | 286 | 1.86 [1.03, 3.34] | 1.15 [0.73, 1.82] | +| D04 | 66–89 | 1.35× | 287 | 1.00 [0.58, 1.71] | 1.00 [0.67, 1.47] | +| D05 | 89–118 | 1.33× | 287 | 1.36 [0.88, 2.12] | 1.19 [0.84, 1.69] | +| D06 | 118–157 | 1.33× | 287 | 0.81 [0.51, 1.28] | 0.78 [0.55, 1.09] | +| D07 | 157–211 | 1.34× | 287 | 1.38 [0.92, 2.06] | 1.16 [0.88, 1.53] | +| D08 | 212–300 | 1.42× | 286 | 0.92 [0.63, 1.33] | 0.96 [0.73, 1.26] | +| D09 | 301–491 | 1.63× | 287 | **1.72 [1.37, 2.16]** | **1.36 [1.16, 1.60]** | +| D10 | 492–8023 | **16.31×** | 287 | **1.47 [1.29, 1.68]** | **1.25 [1.15, 1.36]** | + +**Does RR track the internal range?** Yes, as the dominant pattern. Eight of the ten deciles (D01–D08, each a +genuinely narrow 1.3×–2.9× band) show a CI that includes 1 on both outcomes, with point estimates scattered +on both sides of 1 (0.81 to 1.86) — the shape of noise around no effect, not a consistent direction. The only +two deciles whose CI excludes 1 on both outcomes are D09 and D10. D10 is not a narrow band at all: at 16.31× +internal range it is wider than every tercile ever examined in this document, so a significant RR there is +exactly what "the tercile does not hold size constant" predicts, one level down. D09 (1.63×) is narrower and +its significance does not fit the range story as cleanly — at n=287 per decile with ten bands tested on two +outcomes (20 tests), roughly one false positive at α=0.05 is expected by chance alone even under a true null, +so D09 is reported without a confident causal reading in either direction, not folded into the "it's just +size" story to make the pattern look cleaner than it is. + +**A second, independent check: nearest-token-neighbour matching.** For each of the 717 least-readable-quartile +functions, its closest-toks match from the remaining 2,151 was found (§4's protocol, `nearest_token_match`). +**This check came back unusable, and is reported as such rather than as evidence either way.** Only 46 +distinct functions serve as matches for all 717 quartile members: one function (`parseDeclarator`, 327 toks, +itself FIXED) supplies 403 of the 717 matches (56.2%); a second (`printUsage`, 1,417 toks, also FIXED) +supplies another 126 (17.6%) — together 73.8% of the "matched control" group is two always-fixed functions +repeated hundreds of times, and the mean token gap between a quartile function and its nearest match is +136 +tokens (median 37, max 6,606) — not a close match at all for most of the quartile. The raw number this +produces (matched-control fixed-rate 0.868 vs the quartile's own 0.538, RR 0.62) is **not reported as a +finding**: it is an artifact of how sparse the "rest" population is near the token counts the least-readable +quartile actually occupies, not a size-controlled comparison. That sparsity is itself informative in a +different way — it confirms the quartile sits in a part of the token distribution the "rest" of the +population barely reaches, which is what "the quartile is largely a size cut" predicts — but the RR number +itself is discarded, not used. + +**Restated verdict, using the finest stratification with usable n as decisive.** The pre-registered bands +(§4's Protocol) ask whether the effect vanishes once stratified by size. On deciles narrow enough to +plausibly hold `toks` roughly constant (D01–D08), it does: eight independent CIs on each outcome, none +excluding 1, point estimates with no consistent direction. The two bands that still show a significant +effect are exactly the ones that still carry a wide internal size range (D10 unambiguously; D09 arguably, and +flagged as not cleanly resolved). That is not "keep, narrowed" — this document's second-tier band required +the effect to hold "for size-dominated differences" specifically and to be understood as tracking length; it +is closer to, and is called, this document's **withdraw** band: **the raw and tercile-level separation this +section reported earlier is better explained as a residual size effect, measured in the lens's own units, +than as an independent later-fix signal.** The "keep, narrowed" verdict in the prior revision of this section +is superseded by this one. `--readability`'s ordering does not appear to predict later fixes beyond what its +own dominant token-count component already predicts on its own — which is the same mechanism §3c already +found driving the lens's direction on refactor pairs. Per this document's own precedent (§5, `naminglens.h`), +the correct response to a validation that comes back this way is to say so plainly rather than re-fit the +analysis until it reads more favourably, which is what this subsection does: it withdraws its own prior +verdict in the same document that made it, three commits later, rather than quietly. + +**What would change this again.** A genuine readability-independent signal, if one exists, would need to +show up as a CI excluding 1 on a majority of narrow, size-matched bands — not on the single widest band, and +not on a matching check that turned out to be degenerate. A larger population (a bigger corpus, or a longer +follow-up window) would shrink the decile CIs enough to tell D09 apart from noise, and a working size-matched +control (this repo's own "rest" population is too sparse near the quartile's own token range for 1-NN +matching; a synthetic or cross-repo control pool would not have that gap) would settle the discarded check +above properly instead of leaving it unresolved. + +Re-run: `bench/readability_fixrate_validity.py --bin build/ripwire` (full run, includes deciles and the +matched-control check) or, on an already-written `--out` TSV, `bench/readability_fixrate_validity.py +--load-tsv pop_outcomes.tsv` (re-stratifies in under a second, no re-crawl). + ## 5. Precedent: we have already withdrawn a lens that failed exactly this kind of check This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of