diff --git a/bench/readability_declared_pairs.py b/bench/readability_declared_pairs.py new file mode 100755 index 000000000..9aedc268f --- /dev/null +++ b/bench/readability_declared_pairs.py @@ -0,0 +1,184 @@ +#!/usr/bin/env python3 +r"""readability_declared_pairs.py — proxy (c) and the §3a decomposition of +docs/research/readability-construct-validity.md. + +Proxy (c): commits whose message DECLARES a readability improvement (the Fakhoury ICPC'19 style of ground +truth), scored with exactly §3a's pairing and z comparison (bench/readability_refactor_pairs.py is imported, +not copied, so the two proxies cannot drift apart). The selection rule below was fixed in the paired +document's §3c before this script existed; do not edit it without also saying so there. + +Decomposition: re-runs §3a's own selection and splits every pair's delta-z into its three exact terms — +Halstead volume, token entropy, lines — naming the most negative one as the driver of a wrong-direction pair. + +No LLM judgment anywhere: labels come from commit messages, directions from the shipped binary. + +A measurement harness is a LEDGER, never a red CI gate (bench/ANSWERQUALITY.md): exits 0 whatever it finds. + +Usage: + bench/readability_declared_pairs.py --bin build/ripwire # proxy (c), both arms + bench/readability_declared_pairs.py --bin build/ripwire --decompose # §3a decomposition +""" + +from __future__ import annotations + +import argparse +import re +import shutil +import sys +import tempfile +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import readability_refactor_pairs as rp # noqa: E402 + +# Fixed in §3c's protocol, before any result was computed. +DECLARED_RE = re.compile(r"readab|legib|clarity|clarif|clearer|easier to (read|follow|understand)|self-documenting", re.IGNORECASE) +LENS_NAME_MASK = re.compile(r"--readability|readability\.h|readability_|\(readability\)|\(readability,|readability\)", re.IGNORECASE) + + +def declares_readability(text: str) -> bool: + return DECLARED_RE.search(LENS_NAME_MASK.sub(" ", text)) is not None + + +def changed_lines(sha: str) -> int: + stat = rp.git("show", "--shortstat", "--format=", sha) + return sum(int(tok) for tok in re.findall(r"(\d+) (?:insertion|deletion)s?\(\+?-?\)?", stat)) + + +def declared_commits(max_changed_lines: int, use_body: bool, ref: str = rp.DEFAULT_REF) -> list[tuple[str, str]]: + raw = rp.git("log", "--format=%H\x1f%s\x1f%b\x1e", ref, "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc") + out: list[tuple[str, str]] = [] + for rec in raw.split("\x1e"): + rec = rec.strip("\n") + if not rec.strip(): + continue + sha, _, rest = rec.partition("\x1f") + subject, _, body = rest.partition("\x1f") + if rp.REFACTOR_RE.search(subject): + continue # disjoint from §3a by construction + if not declares_readability(subject + ("\n" + body if use_body else "")): + continue + n = changed_lines(sha) + if n == 0 or n > max_changed_lines: + continue + out.append((sha, subject)) + return out + + +def pairs_for(binary: str, commits: list[tuple[str, str]], scratch: Path) -> list[rp.FnPair]: + pairs: list[rp.FnPair] = [] + for sha, subject in commits: + for path in rp.touched_source_files(sha): + before_text = rp.show_file(f"{sha}^", path) + after_text = rp.show_file(sha, path) + if before_text is None or after_text is None: + continue + basename = Path(path).name + try: + before_rows = rp.score_file(binary, scratch / "before", basename, before_text) + after_rows = rp.score_file(binary, scratch / "after", basename, after_text) + except RuntimeError as exc: + print(f"# skip {sha[:10]} {path}: {exc}", file=sys.stderr) + continue + for name, (lb, tb, volb, entb, pb) in before_rows.items(): + if name not in after_rows: + continue + la, ta, vola, enta, pa = after_rows[name] + if (lb, tb) == (la, ta): + continue + pairs.append(rp.FnPair(sha, subject, path, name, lb, la, tb, ta, volb, vola, entb, enta, pb, pa)) + return pairs + + +def report_direction(label: str, commits: list[tuple[str, str]], pairs: list[rp.FnPair]) -> None: + s = rp.summarize(pairs) + directional = s["improved"] + s["worsened"] + by_commit: dict[str, list[rp.FnPair]] = {} + for p in pairs: + by_commit.setdefault(p.sha, []).append(p) + right = sum(1 for ps in by_commit.values() if sum(p.improved for p in ps) > sum(p.worsened for p in ps)) + wrong = sum(1 for ps in by_commit.values() if sum(p.improved for p in ps) < sum(p.worsened for p in ps)) + deltas = sorted(p.delta for p in pairs) + median = deltas[len(deltas) // 2] if deltas else float("nan") + print(f"== {label}") + print(f"selected commits : {len(commits)} ({len(by_commit)} contributed >=1 pair)") + print(f"directional pairs : {directional} (improved {s['improved']}, worsened {s['worsened']}, tied {s['tied_within_1e-9']})") + frac = s["frac_improved_of_directional"] + print(f"fraction AFTER ranked more readable : {frac:.3f}" if frac == frac else "fraction: n/a") + print(f"per-commit majority right/wrong/split: {right}/{wrong}/{len(by_commit) - right - wrong}") + print(f"mean / median delta-z : {s['mean_z_delta']:.3f} / {median:.3f}") + for sha, subject in commits: + ps = by_commit.get(sha, []) + print(f" {sha[:10]} +{sum(p.improved for p in ps)} -{sum(p.worsened for p in ps)} {subject}") + + +TERMS = ("volume", "entropy", "lines") + + +def contributions(p: rp.FnPair) -> dict[str, float]: + return { + "volume": rp.POSNETT_VOLUME * (p.vol_after - p.vol_before), + "entropy": rp.POSNETT_ENTROPY * (p.ent_after - p.ent_before), + "lines": rp.POSNETT_LINES * (p.lines_after - p.lines_before), + } + + +def report_decomposition(pairs: list[rp.FnPair]) -> None: + for label, subset, pick in (("wrong-direction (delta-z < 0)", [p for p in pairs if p.worsened], min), + ("right-direction (delta-z > 0)", [p for p in pairs if p.improved], max)): + print(f"== {label}: {len(subset)} pairs") + if not subset: + continue + drivers = {t: 0 for t in TERMS} + sums = {t: 0.0 for t in TERMS} + for p in subset: + c = contributions(p) + drivers[pick(TERMS, key=lambda t: c[t])] += 1 + for t in TERMS: + sums[t] += c[t] + for t in TERMS: + print(f" driver={t:<8}: {drivers[t]:>4} ({drivers[t] / len(subset):.1%}) mean contribution {sums[t] / len(subset):+.3f}") + longer = sum(1 for p in subset if p.lines_after > p.lines_before) + shorter = sum(1 for p in subset if p.lines_after < p.lines_before) + more_toks = sum(1 for p in subset if p.toks_after > p.toks_before) + print(f" got longer (dL>0): {longer} shorter: {shorter} same length: {len(subset) - longer - shorter} more tokens: {more_toks}") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=rp.DEFAULT_BIN) + ap.add_argument("--max-changed-lines", type=int, default=400) + ap.add_argument("--decompose", action="store_true", help="decompose §3a's pairs instead of running proxy (c)") + ap.add_argument("--max-commits", type=int, default=80, help="§3a's commit cap, used only with --decompose") + ap.add_argument("--ref", default=rp.DEFAULT_REF, + help=f"walk history from exactly this ref (default {rp.DEFAULT_REF!r}) — never --all") + ap.add_argument("--until", default=None, help="only commits committed at or before this date (git log --until); pins §3a's " + "newest-80 window against refs added later") + args = ap.parse_args() + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin}", file=sys.stderr) + return 2 + if args.until: + plain_git = rp.git + rp.git = lambda *a: plain_git(*a[:1], f"--until={args.until}", *a[1:]) if a and a[0] == "log" else plain_git(*a) + scratch = Path(tempfile.mkdtemp(prefix="rw_declpairs_")) + try: + if args.decompose: + pairs = rp.collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch, args.ref) + s = rp.summarize(pairs) + print(f"§3a regenerated: {s['pairs']} pairs, improved {s['improved']}, worsened {s['worsened']}, tied {s['tied_within_1e-9']}") + report_decomposition(pairs) + else: + for label, use_body in (("PRIMARY arm (subject line)", False), ("SECONDARY arm (subject + body, descriptive only)", True)): + commits = declared_commits(args.max_changed_lines, use_body, args.ref) + pairs = pairs_for(args.bin, commits, scratch) + report_direction(label, commits, pairs) + if not use_body: + report_decomposition(pairs) + finally: + shutil.rmtree(scratch, ignore_errors=True) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/bench/readability_fixrate_validity.py b/bench/readability_fixrate_validity.py new file mode 100644 index 000000000..cb4888c20 --- /dev/null +++ b/bench/readability_fixrate_validity.py @@ -0,0 +1,518 @@ +#!/usr/bin/env python3 +r"""readability_fixrate_validity.py — does `--readability`'s ordering predict anything worth acting on? + +§2 and §3 of docs/research/readability-construct-validity.md validate the ORDER against refactor-commit +direction and self-consistency. Neither asks the question an agent actually relies on when it uses the lens +to pick a refactor target: **does a low readability score predict that a function will need fixing later?** +This script answers that, with no human label and no LLM judge, against the shipped `--readability` lens and +a complexity control the repo's own `--quality-delta` "complexity" kind already trusts (cognitive complexity, +`ccx=`), on the SAME population, with and without controlling for function size. + +Protocol (fixed in this file before any number below existed; see the paired document's new section for the +narrative write-up): + + 1. POPULATION: every `fn`/`method` in `src/**/*.{h,hpp,cpp,cc}` at a pinned CUTOFF commit on `v0.6.2`'s + history (never `--all` — see the instrument-fix note in the paired doc's §3c). The cutoff is chosen to + leave a multi-week, thousand-plus-commit follow-up window to the pinned UNTIL ref (default `v0.6.2`) + so the outcome (see below) has room to happen. Population functions are matched between a real + `--readability` crawl and a real `--metrics` crawl of the SAME extracted tree (`git archive`, not + per-file scratch, so cross-file structure is intact for `--metrics`'s in/out/ccx) by + (path relative to root, function name), with `loc`(metrics) == `lines`(readability) required as a + sanity check; an ambiguous match (same path+name, disagreeing loc, e.g. two overloads) is dropped and + counted, not guessed at. + + 2. EXPOSURE, per function, read at the cutoff: `z` (the exact pre-sigmoid Posnett score, recomputed from + integer `toks=`/`vocab=` exactly as bench/readability_refactor_pairs.py does — lower z = less readable) + and `ccx` (cognitive complexity from `--metrics`, the same metric `src/quality.h`'s "complexity" gate + kind reads — higher ccx = more complex). + + 3. OUTCOME, defined before looking, over the window (CUTOFF, UNTIL]: + - "fix-shaped" commit: subject line (not body — §3c already found body-matching noisy) matches, case + insensitively, `\bfix(e[sd])?\b|\bbug(s|fix(e[sd])?)?\b|\bcrash(e[sd])?\b|\bregression(s)?\b`. + - a function counts as FIXED if any fix-shaped commit in the window has a diff hunk (old-file line + numbers, i.e. against that commit's OWN parent, not the cutoff) that overlaps the function's span + AS MEASURED AT THAT PARENT REVISION (re-scored per commit, so line drift from earlier window + commits cannot misattribute a hunk) — matched back to the population by (basename, function name). + A pure-insertion hunk (old count 0) is treated as touching whichever function(s) contain old-line + `start` or `start+1` (the two lines the insertion sits between). + - a function counts as MODIFIED (the broader arm) under the identical rule, over ALL commits that + touch a src file in the window, fix-shaped or not. Every fix-shaped commit is also a modifying + commit, so both arms are computed in one pass over the window's commits. + + 4. STATISTIC: fix-rate (and modified-rate) in the least-readable quartile (bottom 25% by z) vs the rest, + Wilson-interval per proportion, risk ratio with a Katz log-CI — reported beside the identical statistic + for the highest-ccx quartile vs the rest, on the SAME population, as the trusted-signal baseline. + Repeated within three size terciles TWICE: once by `lines`-at-cutoff, once by `toks`-at-cutoff (Halstead + N, --readability's own toks= — the size measure the lens actually consumes for its volume term, not a + proxy for it), with the least-readable/highest-ccx quartile recomputed WITHIN each tercile. The first + round of this script (see the paired doc's §4) stratified by lines only; a lines-based stratification + does not hold the confound constant, because z tracks the SIGN of the token-count change on 96.0% of + this repo's own refactor pairs (§3c), not the line-count change — a function can gain tokens with no + line change at all. The token-stratified arm is the decisive one for the size-confound question; the + line-stratified arm is kept alongside it because the two disagreeing would itself be informative. + + 5. DECISION BANDS (restated from the paired doc's own pre-registration): the least-readable quartile's + raw fix-rate CI excludes a risk ratio of 1 and the effect does not vanish once stratified by size -> + keep the ordering claim as a weak actionable signal; effect present raw but gone within every size + tercile -> the size confound explains it, disclose and narrow the claim; CI includes 1 raw -> withdraw + the "predicts later fixes" claim outright. The complexity arm is reported for comparison, not as a bar + the lens must clear. + +House rule this script exists under (bench/ANSWERQUALITY.md, bench/BENCHMARK.md): a measurement harness is a +LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say; it is not wired into +test/regression.sh. + +POST-HOC ADDITION (labelled as such, not pre-registered — prompted after seeing the tercile-stratified +result disagree with the line-stratified one): terciles are wide enough that neither one holds the lens's +own size unit (toks) constant within a band, so the script also reports (a) DECILES by toks — ten narrower +bands, to see whether the effect shrinks toward RR=1 as the band narrows, which is what a residual-size +effect would do and a genuine per-token-count-controlled readability effect would not — beside each +decile's own internal token range, since "does RR track the range" is the actual question; and (b) a +nearest-token-neighbour matched control: each least-readable-quartile function against its closest-toks +match from the rest of the population, one-to-one, rather than against a whole band. + +Usage: + bench/readability_fixrate_validity.py --bin build/ripwire + bench/readability_fixrate_validity.py --bin build/ripwire --out pop_outcomes.tsv --json + bench/readability_fixrate_validity.py --load-tsv pop_outcomes.tsv # re-stratify already-computed data + +Deterministic given a fixed git history (pinned CUTOFF/UNTIL refs) and a fixed binary: no randomness anywhere. +""" + +from __future__ import annotations + +import argparse +import bisect +import csv +import json +import math +import re +import shutil +import subprocess +import sys +import tarfile +import tempfile +import xml.etree.ElementTree as ET +from dataclasses import dataclass, field +from pathlib import Path +from typing import Optional + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import readability_refactor_pairs as rp # noqa: E402 (reuses git(), z_score(), DEFAULT_BIN) + +# Chosen to leave a multi-week, thousand-plus-commit follow-up window to v0.6.2 (2286 commits / ~17 days at +# the time this was pinned) — far enough back that a fix-shaped commit in the window cannot be an artifact of +# the cutoff being too recent, close enough that the population is still recognizably today's codebase. +DEFAULT_CUTOFF = "4f5c310c767f6f291c1db2e997b081f35b9a675d" +DEFAULT_UNTIL = "v0.6.2" + +FIX_RE = re.compile(r"\bfix(e[sd])?\b|\bbug(s|fix(e[sd])?)?\b|\bcrash(e[sd])?\b|\bregression(s)?\b", re.IGNORECASE) +HUNK_RE = re.compile(r"^@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@") + + +@dataclass +class FnRow: + path: str # relative to the population root (src/...) + name: str + start_line: int + lines: int + toks: int # Halstead N (operator+operand count) — --readability's own toks=, the exact + # integer z's volume term is computed from; the size measure the lens actually + # consumes, not a proxy for it. + z: float + ccx: int + fixed: bool = False + modified: bool = False + + @property + def basename(self) -> str: + return Path(self.path).name + + @property + def end_line(self) -> int: + return self.start_line + self.lines - 1 + + +def run_xml(binary: str, root: Path, *flags: str) -> ET.Element: + proc = subprocess.run([binary, str(root), *flags], capture_output=True, text=True, timeout=120) + if proc.returncode != 0: + raise RuntimeError(f"{binary} {' '.join(flags)} on {root} exited {proc.returncode}: {proc.stderr.strip()}") + out = proc.stdout + start = out.index("<") + # skip the leading XML comment (legend) to reach the real root element + while out[start:start + 4] == "", start) + 3 + start = out.index("<", start) + return ET.fromstring(out[start:]) + + +def crawl_population(binary: str, src_root: Path) -> dict[tuple[str, str], FnRow]: + """Real crawl of the extracted src/ tree at the cutoff: match --readability rows to --metrics rows by + (path, name), requiring loc==lines to accept the match. Returns {(path, name): FnRow}.""" + read_root = run_xml(binary, src_root, "--readability", "--limit=100000") + met_root = run_xml(binary, src_root, "--metrics", "--top-k=100000") + + readab: dict[tuple[str, str], list[tuple[int, int, int, float]]] = {} # (path,name) -> [(start,lines,toks,z)] + for fn in read_root.findall("fn"): + p = fn.get("p") + path, _, lineno = p.rpartition(":") + toks, vocab = int(fn.get("toks")), int(fn.get("vocab")) + vol_exact = toks * math.log2(vocab) if vocab > 0 else 0.0 + z = rp.z_score(vol_exact, int(fn.get("lines")), float(fn.get("ent"))) + readab.setdefault((path, fn.get("n")), []).append((int(lineno), int(fn.get("lines")), toks, z)) + + metrics: dict[tuple[str, str], list[int]] = {} # (path,name) -> [loc, loc, ...] (one per sc match) + metrics_ccx: dict[tuple[str, str, int], int] = {} + for f in met_root.findall("f"): + path = f.get("p") + for s in f.findall("s"): + if s.get("t") not in ("fn", "method"): + continue + loc = int(s.get("loc")) + key = (path, s.get("n")) + metrics.setdefault(key, []).append(loc) + metrics_ccx[(path, s.get("n"), loc)] = int(s.get("ccx")) + + pop: dict[tuple[str, str], FnRow] = {} + ambiguous = 0 + for key, entries in readab.items(): + locs = metrics.get(key) + if not locs: + continue + for start, lines, toks, z in entries: + if lines not in locs: + continue + ccx = metrics_ccx.get((key[0], key[1], lines)) + if ccx is None: + continue + if key in pop: + ambiguous += 1 + continue + pop[key] = FnRow(path=key[0], name=key[1], start_line=start, lines=lines, toks=toks, z=z, ccx=ccx) + print(f"# population: {len(pop)} matched functions ({ambiguous} ambiguous matches dropped)", file=sys.stderr) + return pop + + +def window_commits(until: str, cutoff: str) -> list[str]: + raw = rp.git("log", "--format=%H", f"{cutoff}..{until}", "--", + "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc") + return [ln for ln in raw.splitlines() if ln.strip()] + + +def parse_hunks(diff_text: str) -> list[tuple[int, int]]: + """[(old_start, old_end_inclusive)] for each hunk; a pure insertion (old count 0) becomes a + zero-width marker (start, start+1) so it can match a function containing either boundary line.""" + out = [] + for line in diff_text.splitlines(): + m = HUNK_RE.match(line) + if not m: + continue + old_start = int(m.group(1)) + old_count = int(m.group(2)) if m.group(2) is not None else 1 + if old_count == 0: + out.append((max(old_start, 1), max(old_start, 1) + 1)) + else: + out.append((old_start, old_start + old_count - 1)) + return out + + +def touched_functions_at_parent(binary: str, scratch: Path, sha: str, path: str) -> list[tuple[str, int, int]]: + """[(name, start, end)] for every function in `path` as it existed at sha^ (the state the diff's old + line numbers are relative to). Runs --readability directly (not rp.score_file) because that helper + drops the start line, which this function needs.""" + text = rp.show_file(f"{sha}^", path) + if text is None: + return [] + scratch_dir = scratch + if scratch_dir.exists(): + shutil.rmtree(scratch_dir) + scratch_dir.mkdir(parents=True) + (scratch_dir / Path(path).name).write_text(text, encoding="utf-8", errors="surrogateescape") + root = run_xml(binary, scratch_dir, "--readability", "--limit=100000") + out = [] + for fn in root.findall("fn"): + p = fn.get("p") + _, _, lineno = p.rpartition(":") + start = int(lineno) + lines = int(fn.get("lines")) + out.append((fn.get("n"), start, start + lines - 1)) + return out + + +def mark_outcomes(binary: str, pop: dict[tuple[str, str], FnRow], commits: list[str], scratch: Path) -> dict: + """One pass over every window commit that touches a src file: for each touched file, resolve the + function spans at that commit's PARENT, intersect against the diff's old-line hunks, and mark any + matching population function (by basename+name) as modified / (if the commit is fix-shaped) fixed.""" + by_basename_name: dict[tuple[str, str], list[FnRow]] = {} + for row in pop.values(): + by_basename_name.setdefault((row.basename, row.name), []).append(row) + + n_fix_commits = 0 + n_touch_commits = 0 + for sha in commits: + subject = rp.git("log", "-1", "--format=%s", sha).strip() + is_fix = bool(FIX_RE.search(subject)) + files = rp.touched_source_files(sha) + if not files: + continue + n_touch_commits += 1 + if is_fix: + n_fix_commits += 1 + for path in files: + diff = rp.git("diff", "-U0", f"{sha}^", sha, "--", path) + hunks = parse_hunks(diff) + if not hunks: + continue + spans = touched_functions_at_parent(binary, scratch, sha, path) + if not spans: + continue + basename = Path(path).name + for name, start, end in spans: + key = (basename, name) + rows = by_basename_name.get(key) + if not rows: + continue + touched = any(not (end < hs or start > he) for hs, he in hunks) + if not touched: + continue + for row in rows: + row.modified = True + if is_fix: + row.fixed = True + return {"fix_commits": n_fix_commits, "touch_commits": n_touch_commits, "window_total": len(commits)} + + +# ---------- statistics (stdlib only) ---------- + +def wilson_interval(k: int, n: int, z: float = 1.959963985) -> tuple[float, float, float]: + if n == 0: + return (float("nan"),) * 3 + p = k / n + denom = 1 + z * z / n + center = (p + z * z / (2 * n)) / denom + half = (z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n))) / denom + return p, max(0.0, center - half), min(1.0, center + half) + + +def risk_ratio_ci(k1: int, n1: int, k0: int, n0: int, z: float = 1.959963985) -> tuple[float, float, float]: + """Katz log-method CI for the risk ratio (group-1 rate / group-0 rate). NaNs out on a zero cell — + disclosed as such by the caller, never silently substituted.""" + if n1 == 0 or n0 == 0 or k1 == 0 or k0 == 0: + return (float("nan"),) * 3 + p1, p0 = k1 / n1, k0 / n0 + rr = p1 / p0 + se = math.sqrt((1 - p1) / (k1) + (1 - p0) / (k0)) if k1 and k0 else float("nan") + lo = rr * math.exp(-z * se) + hi = rr * math.exp(z * se) + return rr, lo, hi + + +def rate_block(flags: list[bool]) -> tuple[int, int, float]: + k = sum(1 for f in flags if f) + n = len(flags) + return k, n, (k / n if n else float("nan")) + + +def quartile_split(rows: list[FnRow], key) -> tuple[list[FnRow], list[FnRow]]: + """Bottom-quartile-by-key vs the rest. `key(row)` ascending; bottom 25% = worst by the study's own + convention (least readable = lowest z; most complex = HIGHEST ccx, so callers pass a negated key).""" + if not rows: + return [], [] + ordered = sorted(rows, key=key) + cut = max(1, round(len(ordered) * 0.25)) + return ordered[:cut], ordered[cut:] + + +def report_arm(label: str, worst: list[FnRow], rest: list[FnRow], outcome: str) -> dict: + wk, wn, wr = rate_block([getattr(r, outcome) for r in worst]) + rk, rn, rr_ = rate_block([getattr(r, outcome) for r in rest]) + _, w_lo, w_hi = wilson_interval(wk, wn) + _, r_lo, r_hi = wilson_interval(rk, rn) + rr, rr_lo, rr_hi = risk_ratio_ci(wk, wn, rk, rn) + return { + "label": label, "outcome": outcome, + "worst_k": wk, "worst_n": wn, "worst_rate": wr, "worst_ci": (w_lo, w_hi), + "rest_k": rk, "rest_n": rn, "rest_rate": rr_, "rest_ci": (r_lo, r_hi), + "risk_ratio": rr, "rr_ci": (rr_lo, rr_hi), + } + + +def size_ntiles(rows: list[FnRow], key=lambda r: r.lines, n: int = 3) -> list[list[FnRow]]: + """N equal-count bands by `key` (default: lines-at-cutoff, 3 = terciles; pass `lambda r: r.toks` for + the token-count measure the lens itself consumes, and n=10 for deciles — a POST-HOC refinement, see + the paired doc's third §4 subsection: terciles are wide enough that a tercile does not hold size + constant on its own, so a real readability effect and a residual-size effect both predict the same + tercile-level RR pattern; only a narrower band distinguishes them).""" + ordered = sorted(rows, key=key) + total = len(ordered) + bounds = [round(i * total / n) for i in range(n + 1)] + return [ordered[bounds[i]:bounds[i + 1]] for i in range(n)] + + +def size_terciles(rows: list[FnRow], key=lambda r: r.lines) -> list[list[FnRow]]: + return size_ntiles(rows, key=key, n=3) + + +def nearest_token_match(pop: list[FnRow], quartile: list[FnRow]) -> list[FnRow]: + """For each function in `quartile`, its nearest neighbour by `toks` among `pop` functions NOT in the + quartile (ties broken by the earlier one in token-sorted order; matching is WITH replacement — a + popular token count can supply more than one match, disclosed at the call site). A cheap alternative + to a fixed-width band: instead of asking "does the effect survive in a band this wide," it asks "does + it survive against a control matched almost exactly on the lens's own size unit.\"""" + quartile_keys = {id(r) for r in quartile} + rest_sorted = sorted((r for r in pop if id(r) not in quartile_keys), key=lambda r: r.toks) + rest_toks = [r.toks for r in rest_sorted] + matches = [] + for q in quartile: + i = bisect.bisect_left(rest_toks, q.toks) + candidates = [j for j in (i - 1, i) if 0 <= j < len(rest_sorted)] + best = min(candidates, key=lambda j: abs(rest_sorted[j].toks - q.toks)) + matches.append(rest_sorted[best]) + return matches + + +def fmt_ci(ci: tuple[float, float]) -> str: + lo, hi = ci + if lo != lo or hi != hi: + return "n/a" + return f"[{lo:.3f}, {hi:.3f}]" + + +def print_report(pop: list[FnRow], meta: dict) -> None: + print(f"population: {len(pop)} functions at cutoff; window: {meta['window_total']} commits touching src, " + f"{meta['touch_commits']} with a matched src diff, {meta['fix_commits']} fix-shaped") + for outcome in ("fixed", "modified"): + print(f"\n=== outcome: {outcome} ===") + worst_z, rest_z = quartile_split(pop, key=lambda r: r.z) + worst_ccx, rest_ccx = quartile_split(pop, key=lambda r: -r.ccx) + for arm in (report_arm("readability (least-readable quartile)", worst_z, rest_z, outcome), + report_arm("complexity (highest-ccx quartile)", worst_ccx, rest_ccx, outcome)): + print(f" {arm['label']}: worst {arm['worst_k']}/{arm['worst_n']} = {arm['worst_rate']:.3f} " + f"{fmt_ci(arm['worst_ci'])} vs rest {arm['rest_k']}/{arm['rest_n']} = {arm['rest_rate']:.3f} " + f"{fmt_ci(arm['rest_ci'])} RR={arm['risk_ratio']:.3f} {fmt_ci(arm['rr_ci'])}") + for strat_label, strat_key, unit_key, unit_name in ( + ("lines-at-cutoff", (lambda r: r.lines), (lambda r: r.lines), "lines"), + ("toks-at-cutoff (Halstead N, --readability's own toks=)", (lambda r: r.toks), (lambda r: r.toks), "toks"), + ): + print(f" -- size-stratified (terciles by {strat_label}, quartile recomputed within each) --") + for i, band in enumerate(size_terciles(pop, key=strat_key), start=1): + wz, rz = quartile_split(band, key=lambda r: r.z) + wc, rc = quartile_split(band, key=lambda r: -r.ccx) + lo_u = min(unit_key(r) for r in band) if band else 0 + hi_u = max(unit_key(r) for r in band) if band else 0 + a = report_arm("readability", wz, rz, outcome) + b = report_arm("complexity", wc, rc, outcome) + print(f" T{i} (n={len(band)}, {unit_name} {lo_u}-{hi_u}):" + f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])} |" + f" complexity RR={b['risk_ratio']:.3f} {fmt_ci(b['rr_ci'])}") + print(" -- POST-HOC (prompted by the tercile result, not pre-registered): " + "DECILES by toks-at-cutoff, quartile recomputed within each decile --") + for i, band in enumerate(size_ntiles(pop, key=lambda r: r.toks, n=10), start=1): + wz, rz = quartile_split(band, key=lambda r: r.z) + a = report_arm("readability", wz, rz, outcome) + lo_t = min(r.toks for r in band) if band else 0 + hi_t = max(r.toks for r in band) if band else 0 + ratio = (hi_t / lo_t) if lo_t > 0 else float("inf") + print(f" D{i:02d} (n={len(band)}, toks {lo_t}-{hi_t}, internal range {ratio:.2f}x):" + f" readab RR={a['risk_ratio']:.3f} {fmt_ci(a['rr_ci'])}") + print(" -- POST-HOC: nearest-token-neighbour matched control " + "(least-readable quartile vs. its 1-NN-by-toks match from the rest) --") + worst_z, _rest_z = quartile_split(pop, key=lambda r: r.z) + matched = nearest_token_match(pop, worst_z) + wk, wn, wr = rate_block([getattr(r, outcome) for r in worst_z]) + mk, mn, mr = rate_block([getattr(r, outcome) for r in matched]) + _, w_lo, w_hi = wilson_interval(wk, wn) + _, m_lo, m_hi = wilson_interval(mk, mn) + rr, rr_lo, rr_hi = risk_ratio_ci(wk, wn, mk, mn) + mean_gap = (sum(r.toks for r in worst_z) - sum(r.toks for r in matched)) / len(worst_z) if worst_z else 0.0 + print(f" least-readable quartile {wk}/{wn} = {wr:.3f} {fmt_ci((w_lo, w_hi))} vs matched control " + f"{mk}/{mn} = {mr:.3f} {fmt_ci((m_lo, m_hi))} RR={rr:.3f} {fmt_ci((rr_lo, rr_hi))}" + f" (mean token gap quartile-minus-match: {mean_gap:+.1f})") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=rp.DEFAULT_BIN) + ap.add_argument("--cutoff", default=DEFAULT_CUTOFF) + ap.add_argument("--until", default=DEFAULT_UNTIL) + ap.add_argument("--out", default=None, help="write the per-function TSV here") + ap.add_argument("--json", action="store_true") + ap.add_argument("--scratch", default=None) + ap.add_argument("--load-tsv", default=None, + help="skip the population crawl and outcome-marking pass (the slow ~15-minute step) " + "and re-load a previously written --out TSV instead — for re-stratifying " + "(--strata, deciles, the matched control) on data already computed. The window/" + "fix-commit counts in the report line are not available from a TSV alone and " + "print as 'n/a'.") + args = ap.parse_args() + + if args.load_tsv: + pop = [] + with open(args.load_tsv, encoding="utf-8") as f: + for row in csv.DictReader(f, delimiter="\t"): + pop.append(FnRow(path=row["path"], name=row["name"], start_line=int(row["start_line"]), + lines=int(row["lines"]), toks=int(row["toks"]), z=float(row["z"]), + ccx=int(row["ccx"]), fixed=bool(int(row["fixed"])), + modified=bool(int(row["modified"])))) + meta = {"window_total": "n/a (--load-tsv)", "touch_commits": "n/a", "fix_commits": "n/a"} + else: + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin}", file=sys.stderr) + return 2 + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_fixrate_")) + try: + pop_src = scratch / "population_src" + pop_src.mkdir(parents=True) + archive = subprocess.run(["git", "archive", args.cutoff, "--", "src"], cwd=str(rp.ROOT), + capture_output=True, timeout=60) + if archive.returncode != 0: + print(f"error: git archive {args.cutoff} -- src failed: {archive.stderr.decode()}", file=sys.stderr) + return 2 + with tempfile.NamedTemporaryFile(suffix=".tar") as tf: + tf.write(archive.stdout) + tf.flush() + with tarfile.open(tf.name) as tar: + tar.extractall(pop_src) + + pop_map = crawl_population(args.bin, pop_src) + commits = window_commits(args.until, args.cutoff) + meta = mark_outcomes(args.bin, pop_map, commits, scratch / "mark_scratch") + meta["window_total"] = len(commits) + pop = list(pop_map.values()) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) + + if args.out: + with open(args.out, "w", encoding="utf-8") as f: + f.write("path\tname\tstart_line\tlines\ttoks\tz\tccx\tfixed\tmodified\n") + for r in pop: + f.write(f"{r.path}\t{r.name}\t{r.start_line}\t{r.lines}\t{r.toks}\t{r.z:.6f}\t{r.ccx}\t{int(r.fixed)}\t{int(r.modified)}\n") + print(f"# wrote {len(pop)} rows to {args.out}", file=sys.stderr) + + if args.json: + def arm_json(rows, key): + worst, rest = quartile_split(rows, key=key) + return {o: report_arm("x", worst, rest, o) for o in ("fixed", "modified")} + def strat_json(strat_key): + return [ + {"n": len(band), + "readability": arm_json(band, lambda r: r.z), + "complexity": arm_json(band, lambda r: -r.ccx)} + for band in size_terciles(pop, key=strat_key) + ] + print(json.dumps({ + "meta": meta, + "n_population": len(pop), + "readability": arm_json(pop, lambda r: r.z), + "complexity": arm_json(pop, lambda r: -r.ccx), + "stratified_by_lines": strat_json(lambda r: r.lines), + "stratified_by_toks": strat_json(lambda r: r.toks), + }, indent=2, default=str)) + else: + print_report(pop, meta) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/bench/readability_refactor_pairs.py b/bench/readability_refactor_pairs.py new file mode 100755 index 000000000..f9cd0e647 --- /dev/null +++ b/bench/readability_refactor_pairs.py @@ -0,0 +1,292 @@ +#!/usr/bin/env python3 +r"""readability_refactor_pairs.py — proxy (a) of docs/research/readability-construct-validity.md: without +any human label, does `--readability`'s Posnett rank move the direction a refactor/simplify/cleanup commit +message says it should? + +For each commit in this repo's history whose subject matches /refactor|simplify|clean(\s|-)?up/i and that +touches a bounded number of C/C++ source lines, this script extracts each touched .h/.cpp file at the +commit and at its parent, scores every function in each version with the real `--readability` lens (via a +single-file scratch directory, so the lens's own tokenizer and Posnett fit run unmodified), and matches +functions by (file basename, name) that are present in both versions with a DIFFERENT token shape (a +same-shape match means the diff did not touch that specific function's body and is not evidence either +way). It reports what fraction of genuinely-changed functions the lens ranks MORE readable after a commit +whose author called it a refactor — a cheap, label-free stand-in for "does the ranking move the direction a +human editor intended." + +This is a proxy, not ground truth: a commit message is not a readability judgement, and "refactor" commits +sometimes trade readability for something else (performance, a new abstraction boundary) the author valued +more. Read the numbers this way, and see the "what we would like help with" section of the paired doc. + +House rule this script exists under (bench/ANSWERQUALITY.md, bench/BENCHMARK.md): a measurement harness is +a LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say; it is not wired +into test/regression.sh. + +Usage: + bench/readability_refactor_pairs.py # default: this repo, 60 commits, src/ + bench/readability_refactor_pairs.py --max-commits 150 --out /tmp/pairs.tsv + RIPWIRE_BIN=asan/ripwire bench/readability_refactor_pairs.py --json > pairs.json + +Deterministic given a fixed git history and a fixed binary: git log ordering is chronological, the lens +itself is deterministic (docs/ARCHITECTURE.md), and this script does not thread. +""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import re +import shutil +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path +from typing import Optional + +ROOT = Path(__file__).resolve().parent.parent +DEFAULT_BIN = os.environ.get("RIPWIRE_BIN", str(ROOT / "build" / "ripwire")) + +# The population must be pinned to ONE immutable ref, never `--all`: this clone's shared .git carries every +# worktree's branches, so `--all` makes the candidate-commit population — and therefore the published +# fraction — move whenever any unrelated lane is pushed. That is an instrument defect, not a measurement +# (docs/research/readability-construct-validity.md §3a). v0.6.2 is the tag the shipped scoring binary was +# built from; override with --ref for a different pinned population, but never pass --all here. +DEFAULT_REF = "v0.6.2" + +REFACTOR_RE = re.compile(r"refactor|simplify|clean(\s|-)?up", re.IGNORECASE) +SOURCE_EXT = {".h", ".hpp", ".cc", ".cpp", ".cxx"} + +# The Posnett coefficients, copied verbatim from src/readability.h (kPosnettIntercept/Volume/Lines/Entropy) +# so this script can recompute the pre-sigmoid z on FULL precision instead of comparing the CLI's own +# 3-decimal, sigmoid-saturating `posnett=` attribute. z is monotonic in posnett (sigmoid is monotonic), so +# ranking by z ranks identically to ranking by posnett, without the display truncation or the saturation +# that makes several distinct functions all print posnett="0.000" (readability.h's own legend text says +# so). toks= and vocab= are exact integers in the XML, so vol = toks*log2(vocab) is recovered EXACTLY — +# only ent= (2 decimals as printed) stays at display precision, since the per-token frequency table itself +# is not exposed by the verb. +POSNETT_INTERCEPT = 8.87 +POSNETT_VOLUME = -0.033 +POSNETT_LINES = 0.40 +POSNETT_ENTROPY = -1.5 + + +def z_score(vol: float, lines: int, ent: float) -> float: + return POSNETT_INTERCEPT + POSNETT_VOLUME * vol + POSNETT_LINES * lines + POSNETT_ENTROPY * ent + + +@dataclass +class FnPair: + sha: str + subject: str + path: str + name: str + lines_before: int + lines_after: int + toks_before: int + toks_after: int + vol_before: float + vol_after: float + ent_before: float + ent_after: float + posnett_before: float # the CLI's own displayed value — kept for reference, not for comparison + posnett_after: float + + @property + def z_before(self) -> float: + return z_score(self.vol_before, self.lines_before, self.ent_before) + + @property + def z_after(self) -> float: + return z_score(self.vol_after, self.lines_after, self.ent_after) + + @property + def delta(self) -> float: + return self.z_after - self.z_before + + @property + def improved(self) -> bool: + return self.z_after > self.z_before + 1e-9 + + @property + def worsened(self) -> bool: + return self.z_after < self.z_before - 1e-9 + + +def git(*args: str) -> str: + proc = subprocess.run(["git", *args], cwd=str(ROOT), capture_output=True, text=True) + if proc.returncode != 0: + raise RuntimeError(f"git {' '.join(args)} failed: {proc.stderr.strip()}") + return proc.stdout + + +def candidate_commits(max_commits: int, max_changed_lines: int, ref: str = DEFAULT_REF) -> list[tuple[str, str]]: + """(sha, subject) pairs, newest first, whose subject reads as a refactor/simplify/cleanup and whose + total changed-line count over the whole commit is small enough that attributing a per-function score + delta to "the refactor" is defensible rather than noise from an unrelated bulk edit riding along. + + Walks exactly ONE immutable ref (default the v0.6.2 tag), never `--all`: see DEFAULT_REF's comment.""" + raw = git( + "log", "--format=%H\x1f%s", ref, "-i", + "--grep=refactor", "--grep=simplify", "--grep=clean up", "--grep=cleanup", + "--", "src/*.h", "src/*.hpp", "src/*.cpp", "src/*.cc", + ) + out: list[tuple[str, str]] = [] + for line in raw.splitlines(): + if not line.strip(): + continue + sha, _, subject = line.partition("\x1f") + if not REFACTOR_RE.search(subject): + continue + stat = git("show", "--shortstat", "--format=", sha) + m = re.search(r"(\d+) insertion.*?(\d+) deletion|(\d+) insertion|(\d+) deletion", stat) + changed = 0 + for tok in re.findall(r"(\d+) (?:insertion|deletion)s?\(\+?-?\)?", stat): + changed += int(tok) + if changed == 0 or changed > max_changed_lines: + continue + out.append((sha, subject)) + if len(out) >= max_commits: + break + return out + + +def touched_source_files(sha: str) -> list[str]: + raw = git("show", "--name-only", "--format=", sha) + return [p for p in raw.splitlines() if p and Path(p).suffix in SOURCE_EXT and p.startswith("src/")] + + +def show_file(rev: str, path: str) -> Optional[str]: + proc = subprocess.run(["git", "show", f"{rev}:{path}"], cwd=str(ROOT), capture_output=True, text=True) + if proc.returncode != 0: + return None + return proc.stdout + + +def score_file(binary: str, scratch_dir: Path, basename: str, text: str) -> dict[str, tuple]: + """Write `text` as the sole file in an otherwise-empty directory, run --readability over it, and + return {function_name: (lines, toks, vol_exact, ent, posnett)}. A single-file directory is enough — + the lens is per-function and needs no cross-file resolution (readability.h's own header note).""" + if scratch_dir.exists(): + shutil.rmtree(scratch_dir) + scratch_dir.mkdir(parents=True) + (scratch_dir / basename).write_text(text, encoding="utf-8", errors="surrogateescape") + proc = subprocess.run( + [binary, str(scratch_dir), "--readability", "--limit=100000"], + capture_output=True, text=True, timeout=60, + ) + if proc.returncode != 0: + raise RuntimeError(f"--readability exited {proc.returncode} on {scratch_dir}: {proc.stderr.strip()}") + root = ET.fromstring(proc.stdout) + rows: dict[str, tuple] = {} + for fn in root.findall("fn"): + name = fn.get("n") + # Same-named overload/specialization collisions are rare in one file; keep the first, since a + # scratch single-file rescoring only needs A representative match, not perfect overload resolution. + if name not in rows: + toks = int(fn.get("toks")) + vocab = int(fn.get("vocab")) + vol_exact = toks * math.log2(vocab) if vocab > 0 else 0.0 # readability.h's own formula, integer inputs + rows[name] = (int(fn.get("lines")), toks, vol_exact, float(fn.get("ent")), float(fn.get("posnett"))) + return rows + + +def collect_pairs(binary: str, max_commits: int, max_changed_lines: int, scratch: Path, ref: str = DEFAULT_REF) -> list[FnPair]: + pairs: list[FnPair] = [] + commits = candidate_commits(max_commits, max_changed_lines, ref) + print(f"# {len(commits)} candidate refactor/simplify/cleanup commits (max_changed_lines={max_changed_lines}, ref={ref})", file=sys.stderr) + before_dir = scratch / "before" + after_dir = scratch / "after" + for sha, subject in commits: + for path in touched_source_files(sha): + before_text = show_file(f"{sha}^", path) + after_text = show_file(sha, path) + if before_text is None or after_text is None: + continue # file added or deleted in this commit — no pair to compare + basename = Path(path).name + try: + before_rows = score_file(binary, before_dir, basename, before_text) + after_rows = score_file(binary, after_dir, basename, after_text) + except RuntimeError as exc: + print(f"# skip {sha[:10]} {path}: {exc}", file=sys.stderr) + continue + for name, (lb, tb, volb, entb, pb) in before_rows.items(): + if name not in after_rows: + continue + la, ta, vola, enta, pa = after_rows[name] + if (lb, tb) == (la, ta): + continue # identical shape — this function's body was not touched by the diff + pairs.append(FnPair(sha, subject, path, name, lb, la, tb, ta, volb, vola, entb, enta, pb, pa)) + return pairs + + +def summarize(pairs: list[FnPair]) -> dict: + improved = sum(1 for p in pairs if p.improved) + worsened = sum(1 for p in pairs if p.worsened) + tied = len(pairs) - improved - worsened + n_directional = improved + worsened + frac_improved = improved / n_directional if n_directional else float("nan") + mean_delta = sum(p.delta for p in pairs) / len(pairs) if pairs else float("nan") + return { + "pairs": len(pairs), + "improved": improved, + "worsened": worsened, + "tied_within_1e-9": tied, + "frac_improved_of_directional": frac_improved, + "mean_z_delta": mean_delta, + } + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=DEFAULT_BIN) + ap.add_argument("--max-commits", type=int, default=60) + ap.add_argument("--max-changed-lines", type=int, default=400, + help="skip commits whose total insertion+deletion count exceeds this (default 400) — " + "keeps the compared functions attributable to the named refactor") + ap.add_argument("--ref", default=DEFAULT_REF, + help=f"walk history from exactly this ref (default {DEFAULT_REF!r}) — never --all, " + "which makes the population move whenever any unrelated branch is pushed to a " + "shared .git") + ap.add_argument("--out", default=None, help="write the per-pair TSV here (default: stdout table only)") + ap.add_argument("--json", action="store_true", help="print the summary as JSON instead of a table") + ap.add_argument("--scratch", default=None, help="scratch directory (default: a fresh temp dir)") + args = ap.parse_args() + + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin} — build it first (see CLAUDE.md)", file=sys.stderr) + return 2 + + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_readpairs_")) + try: + pairs = collect_pairs(args.bin, args.max_commits, args.max_changed_lines, scratch, args.ref) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) + + summary = summarize(pairs) + + if args.out: + with open(args.out, "w", encoding="utf-8") as f: + f.write("sha\tsubject\tpath\tname\tz_before\tz_after\tz_delta\tposnett_before\tposnett_after\tlines_before\tlines_after\ttoks_before\ttoks_after\n") + for p in pairs: + f.write(f"{p.sha}\t{p.subject}\t{p.path}\t{p.name}\t{p.z_before:.6f}\t{p.z_after:.6f}\t{p.delta:.6f}\t{p.posnett_before:.3f}\t{p.posnett_after:.3f}\t{p.lines_before}\t{p.lines_after}\t{p.toks_before}\t{p.toks_after}\n") + print(f"# wrote {len(pairs)} pairs to {args.out}", file=sys.stderr) + + if args.json: + print(json.dumps(summary, indent=2)) + else: + print(f"pairs (changed-function before/after) : {summary['pairs']}") + print(f"lens ranks AFTER more readable (z increased) : {summary['improved']}") + print(f"lens ranks AFTER less readable (z decreased) : {summary['worsened']}") + print(f"exact tie (z unchanged, to 1e-9) : {summary['tied_within_1e-9']}") + frac = summary["frac_improved_of_directional"] + print(f"fraction improved, of the directional pairs : {frac:.3f}" if frac == frac else "fraction improved: n/a (no directional pairs)") + print(f"mean z delta (after - before; +delta=more readable) : {summary['mean_z_delta']:.4f}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/bench/readability_self_consistency.py b/bench/readability_self_consistency.py new file mode 100755 index 000000000..1747d3aba --- /dev/null +++ b/bench/readability_self_consistency.py @@ -0,0 +1,354 @@ +#!/usr/bin/env python3 +r"""readability_self_consistency.py — proxy (b) of docs/research/readability-construct-validity.md: without +any human label, does `--readability` agree with ITSELF across two mechanical rewrites that should not +change how readable a function is — renaming one local to a fresh, non-colliding name, and reordering two +adjacent, mutually-independent simple statements? + +Method: sample functions from this repo's own `src` at percentile positions across the lens's own ranked +list (least-readable-first), so the sample spans the whole distribution rather than only the worst decile +--readability itself prints by default. For each sampled function, re-score its OWN whole file unmodified +in a single-file scratch directory (so the comparison baseline takes the identical measurement path the +mutant does — no cross-file ingest differences to explain away), apply one mutation, re-score the mutated +whole file, and compare. + +WHAT THIS TEST ACTUALLY MEASURES, STATED PLAINLY. Halstead volume and Shannon entropy are functions of the +MULTISET of token (class, frequency) pairs, not of token identity or statement order — readability.h's own +determinism note says the entropy sum runs over a token-text-sorted vector precisely because order must not +reach the output. So for a rename to a fresh (non-colliding) identifier, or a reorder of two independent +statements, ZERO change in `vol=`/`ent=`/`posnett=` is what the formula predicts mathematically, not a +finding this script discovers. A pass here mostly certifies IMPLEMENTATION correctness (no stray +non-determinism, no order leak) rather than any claim about human-perceived readability invariance — and it +is honest about that limit rather than overselling a trivial pass as construct validity. The genuinely +informative reading is the FAILURE case: any pair that does NOT tie exactly is either an extraction/mutation +bug in this harness or a real order/identity sensitivity in the lens worth filing. The other honest reading +is what this test structurally CANNOT catch: because the formula is blind to identifier length and to +statement order by construction, it also cannot reward `numberOfActiveConnections` over `n`, or a +better-ordered proof over a worse one — see the "what we would like help with" section of the paired doc. + +House rule this script exists under (bench/ANSWERQUALITY.md, bench/BENCHMARK.md): a measurement harness is +a LEDGER, never a red CI gate. It reports numbers and exits 0 regardless of what they say. + +Usage: + bench/readability_self_consistency.py # default: this repo's src/, 30 samples + bench/readability_self_consistency.py --samples 60 --seed 7 + RIPWIRE_BIN=asan/ripwire bench/readability_self_consistency.py --json > consistency.json +""" + +from __future__ import annotations + +import argparse +import json +import os +import random +import re +import shutil +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path +from typing import Optional + +ROOT = Path(__file__).resolve().parent.parent +DEFAULT_BIN = os.environ.get("RIPWIRE_BIN", str(ROOT / "build" / "ripwire")) +DEFAULT_CORPUS = "src" + +# A fixed, deliberately weird suffix: astronomically unlikely to already occur as a real identifier in this +# tree, so a rename to `_mutrn7k` cannot collide with an existing token and silently change vocab=. +RENAME_SUFFIX = "_mutrn7k" + +CPP_KEYWORDS = { + "if", "else", "for", "while", "do", "switch", "case", "default", "break", "continue", "return", + "const", "static", "inline", "constexpr", "auto", "void", "int", "bool", "char", "double", "float", + "struct", "class", "namespace", "using", "typename", "template", "public", "private", "protected", + "true", "false", "nullptr", "new", "delete", "this", "sizeof", "noexcept", "override", "virtual", + "std", "size_t", "uint32_t", "uint64_t", "int32_t", "int64_t", "std::size_t", +} + +IDENT_RE = re.compile(r"\b[a-z_][a-zA-Z0-9_]*\b") +IDENT_ANY_RE = re.compile(r"[A-Za-z_]\w*") +# A conservative "simple statement" shape: one line, ending `;`, no braces (so it is not secretly a whole +# block), with exactly one PLAIN `=` — not `==`/`!=`/`<=`/`>=` and not a compound `+=`/`-=`/… . Matches both +# a bare reassignment (`x = y + z;`) and a one-line typed declaration (`const Foo& x = expr;`), which this +# house style uses constantly (CONTRIBUTING.md Allman style, one statement per line). +SIMPLE_STMT_RE = re.compile(r"^(?P\s*)(?P[^{};]+);\s*$") +COMPOUND_EQ_PREV = set("=!<>+-*/%&|^") + + +def parse_simple_stmt(line: str) -> Optional[tuple[str, str, str]]: + """(indent, lhs_variable_name, rhs_text) for a line this script is willing to reorder, or None. The LHS + name is the LAST identifier immediately before the plain `=` — for a typed declaration that is the + variable, not the type.""" + m = SIMPLE_STMT_RE.match(line) + if not m: + return None + body = m.group("body") + plain_eqs = [i for i, ch in enumerate(body) + if ch == "=" and (i == 0 or body[i - 1] not in COMPOUND_EQ_PREV) and body[i + 1:i + 2] != "="] + if len(plain_eqs) != 1: + return None + i = plain_eqs[0] + lhs_part, rhs_part = body[:i].rstrip(), body[i + 1:].strip() + if "(" in lhs_part or "[" in lhs_part or "->" in lhs_part: + return None # a call, subscript or member target — too easy to get independence wrong + idents = IDENT_ANY_RE.findall(lhs_part) + if not idents or idents[-1] in CPP_KEYWORDS: + return None + return m.group("indent"), idents[-1], rhs_part + + +@dataclass +class FnRow: + path: str + line: int + name: str + lines: int + posnett: float + rank_index: int # position in the lens's own least-readable-first ordering + + +def run_readability(binary: str, target: str, cwd: Optional[Path] = None) -> list[FnRow]: + proc = subprocess.run( + [binary, target, "--readability", "--limit=100000"], + capture_output=True, text=True, timeout=120, cwd=str(cwd) if cwd else None, + ) + if proc.returncode != 0: + raise RuntimeError(f"--readability exited {proc.returncode} on {target}: {proc.stderr.strip()}") + root = ET.fromstring(proc.stdout) + rows = [] + for i, fn in enumerate(root.findall("fn")): + p, _, ln = fn.get("p").rpartition(":") + rows.append(FnRow(p, int(ln), fn.get("n"), int(fn.get("lines")), float(fn.get("posnett")), i)) + return rows + + +def score_single_file(binary: str, scratch_dir: Path, basename: str, text: str) -> dict[str, dict]: + if scratch_dir.exists(): + shutil.rmtree(scratch_dir) + scratch_dir.mkdir(parents=True) + (scratch_dir / basename).write_text(text, encoding="utf-8", errors="surrogateescape") + proc = subprocess.run( + [binary, str(scratch_dir), "--readability", "--limit=100000"], + capture_output=True, text=True, timeout=60, + ) + if proc.returncode != 0: + raise RuntimeError(f"--readability exited {proc.returncode} on {scratch_dir}: {proc.stderr.strip()}") + root = ET.fromstring(proc.stdout) + out: dict[str, dict] = {} + for fn in root.findall("fn"): + name = fn.get("n") + if name not in out: + out[name] = { + "lines": int(fn.get("lines")), "toks": int(fn.get("toks")), "vocab": int(fn.get("vocab")), + "vol": float(fn.get("vol")), "ent": float(fn.get("ent")), "posnett": float(fn.get("posnett")), + } + return out + + +def stratified_sample(rows: list[FnRow], n: int, rng: random.Random) -> list[FnRow]: + """n rows spread across the WHOLE ranked distribution (percentile bins with one jittered pick each), not + just the worst decile the verb shows by default — a consistency check only worth trusting if it covers + functions the lens currently calls readable as well as ones it calls unreadable.""" + if len(rows) <= n: + return list(rows) + picked = [] + bin_size = len(rows) / n + for i in range(n): + lo = int(i * bin_size) + hi = max(lo + 1, int((i + 1) * bin_size)) + hi = min(hi, len(rows)) + picked.append(rows[rng.randrange(lo, hi)]) + return picked + + +def extract_span(file_lines: list[str], start_line: int, span: int) -> Optional[tuple[int, int]]: + """1-indexed inclusive [start_line, start_line+span-1], clamped; None if out of range.""" + lo = start_line - 1 + hi = lo + span + if lo < 0 or hi > len(file_lines) or lo >= hi: + return None + return lo, hi + + +def try_rename(span_lines: list[str], fn_name: str) -> Optional[tuple[list[str], str]]: + """Pick a lowercase identifier appearing >=2 times in the span, not a keyword, not the function's own + name, not already ending in RENAME_SUFFIX, whole-word-replace every occurrence. Returns (new_lines, + renamed_identifier) or None if no eligible candidate exists.""" + text = "\n".join(span_lines) + counts: dict[str, int] = {} + for m in IDENT_RE.finditer(text): + tok = m.group(0) + if tok in CPP_KEYWORDS or tok == fn_name or len(tok) < 2 or tok.endswith(RENAME_SUFFIX): + continue + counts[tok] = counts.get(tok, 0) + 1 + candidates = [t for t, c in counts.items() if c >= 2] + if not candidates: + return None + # Deterministic pick: the most frequent candidate, ties broken lexically — reproducible across runs. + candidates.sort(key=lambda t: (-counts[t], t)) + victim = candidates[0] + new_name = victim + RENAME_SUFFIX + if new_name in text: + return None # pathological collision; skip rather than risk a silent vocab change + pattern = re.compile(r"\b" + re.escape(victim) + r"\b") + new_text = pattern.sub(new_name, text) + return new_text.split("\n"), victim + + +def try_reorder(span_lines: list[str]) -> Optional[tuple[list[str], int]]: + """Find the first adjacent pair of simple statement lines (parse_simple_stmt), same indentation, + different LHS name, where neither RHS mentions the other's LHS as a whole word (the independence + heuristic) — and swap the two full lines.""" + for i in range(len(span_lines) - 1): + p1 = parse_simple_stmt(span_lines[i]) + p2 = parse_simple_stmt(span_lines[i + 1]) + if not p1 or not p2: + continue + indent1, lhs1, rhs1 = p1 + indent2, lhs2, rhs2 = p2 + if indent1 != indent2 or lhs1 == lhs2: + continue + if re.search(r"\b" + re.escape(lhs1) + r"\b", rhs2) or re.search(r"\b" + re.escape(lhs2) + r"\b", rhs1): + continue + out = list(span_lines) + out[i], out[i + 1] = out[i + 1], out[i] + return out, i + return None + + +@dataclass +class MutationResult: + path: str + name: str + kind: str # "rename" | "reorder" + detail: str + baseline: dict + mutant: dict + + def ties(self) -> bool: + return (abs(self.baseline["vol"] - self.mutant["vol"]) < 1e-6 + and abs(self.baseline["ent"] - self.mutant["ent"]) < 1e-6 + and self.baseline["posnett"] == self.mutant["posnett"]) + + +def run(binary: str, corpus: str, n_samples: int, seed: int, scratch: Path) -> list[MutationResult]: + rng = random.Random(seed) + all_rows = run_readability(binary, corpus, cwd=ROOT) + print(f"# {len(all_rows)} functions measured in {corpus}; sampling {n_samples} across the ranked distribution", file=sys.stderr) + sample = stratified_sample(all_rows, n_samples, rng) + + results: list[MutationResult] = [] + baseline_dir = scratch / "baseline" + mutant_dir = scratch / "mutant" + for row in sample: + # `p=` in the lens's own output is relative to the scanned root (its legend says so verbatim); + # rejoin against the corpus argument this run used to get a real path. + abspath = ROOT / corpus / row.path + if not abspath.is_file(): + continue + whole = abspath.read_text(encoding="utf-8", errors="surrogateescape") + file_lines = whole.split("\n") + span = extract_span(file_lines, row.line, row.lines) + if span is None: + continue + lo, hi = span + span_lines = file_lines[lo:hi] + span_text = "\n".join(span_lines) + if row.name not in span_text: + print(f"# skip {row.path}:{row.line} {row.name} — extracted span does not mention its own name (line-mapping drift)", file=sys.stderr) + continue + basename = Path(row.path).name + + try: + baseline_scores = score_single_file(binary, baseline_dir, basename, whole) + except RuntimeError as exc: + print(f"# skip {row.path} baseline: {exc}", file=sys.stderr) + continue + if row.name not in baseline_scores: + continue + base_row = baseline_scores[row.name] + + renamed = try_rename(span_lines, row.name) + if renamed is not None: + new_span, victim = renamed + new_lines = file_lines[:lo] + new_span + file_lines[hi:] + mutant_text = "\n".join(new_lines) + try: + mutant_scores = score_single_file(binary, mutant_dir, basename, mutant_text) + if row.name in mutant_scores: + results.append(MutationResult(row.path, row.name, "rename", f"{victim} -> {victim}{RENAME_SUFFIX}", + base_row, mutant_scores[row.name])) + except RuntimeError as exc: + print(f"# skip {row.path} rename: {exc}", file=sys.stderr) + + reordered = try_reorder(span_lines) + if reordered is not None: + new_span, at = reordered + new_lines = file_lines[:lo] + new_span + file_lines[hi:] + mutant_text = "\n".join(new_lines) + try: + mutant_scores = score_single_file(binary, mutant_dir, basename, mutant_text) + if row.name in mutant_scores: + results.append(MutationResult(row.path, row.name, "reorder", f"swap span-local lines {at},{at+1}", + base_row, mutant_scores[row.name])) + except RuntimeError as exc: + print(f"# skip {row.path} reorder: {exc}", file=sys.stderr) + + return results + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--bin", default=DEFAULT_BIN) + ap.add_argument("--corpus", default=DEFAULT_CORPUS) + ap.add_argument("--samples", type=int, default=30) + ap.add_argument("--seed", type=int, default=20260918) + ap.add_argument("--out", default=None) + ap.add_argument("--json", action="store_true") + ap.add_argument("--scratch", default=None) + args = ap.parse_args() + + if not Path(args.bin).is_file(): + print(f"error: ripwire binary not found at {args.bin} — build it first (see CLAUDE.md)", file=sys.stderr) + return 2 + + scratch = Path(args.scratch) if args.scratch else Path(tempfile.mkdtemp(prefix="rw_readconsist_")) + try: + results = run(args.bin, args.corpus, args.samples, args.seed, scratch) + finally: + if not args.scratch: + shutil.rmtree(scratch, ignore_errors=True) + + by_kind: dict[str, list[MutationResult]] = {"rename": [], "reorder": []} + for r in results: + by_kind[r.kind].append(r) + + summary = {} + for kind, rs in by_kind.items(): + ties = sum(1 for r in rs if r.ties()) + summary[kind] = {"attempted": len(rs), "exact_tie": ties, "diverged": len(rs) - ties} + + if args.out: + with open(args.out, "w", encoding="utf-8") as f: + f.write("kind\tpath\tname\tdetail\ttie\tvol_before\tvol_after\tent_before\tent_after\tposnett_before\tposnett_after\n") + for r in results: + f.write(f"{r.kind}\t{r.path}\t{r.name}\t{r.detail}\t{int(r.ties())}\t{r.baseline['vol']:.4f}\t{r.mutant['vol']:.4f}\t{r.baseline['ent']:.4f}\t{r.mutant['ent']:.4f}\t{r.baseline['posnett']:.3f}\t{r.mutant['posnett']:.3f}\n") + print(f"# wrote {len(results)} mutation results to {args.out}", file=sys.stderr) + + if args.json: + print(json.dumps(summary, indent=2)) + else: + for kind, s in summary.items(): + print(f"{kind:8s} attempted={s['attempted']:3d} exact_tie={s['exact_tie']:3d} diverged={s['diverged']:3d}") + for r in results: + if not r.ties(): + print(f" DIVERGED [{r.kind}] {r.path} {r.name} ({r.detail}): " + f"vol {r.baseline['vol']:.2f}->{r.mutant['vol']:.2f} " + f"ent {r.baseline['ent']:.2f}->{r.mutant['ent']:.2f} " + f"posnett {r.baseline['posnett']:.3f}->{r.mutant['posnett']:.3f}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/README.md b/docs/README.md index a9ba460b8..4d459a24f 100644 --- a/docs/README.md +++ b/docs/README.md @@ -1,6 +1,6 @@ # ripwire documentation -Sixteen entries, each written for one reader. Start with the row that matches why you are here. +Twenty-one entries, each written for one reader. Start with the row that matches why you are here. | File | Who it is for | What it answers | | --- | --- | --- | @@ -22,6 +22,7 @@ Sixteen entries, each written for one reader. Start with the row that matches wh | **[`gatecount_build.py`](gatecount_build.py)** | Maintainers, and anyone adding a gate | The generator behind the **published gate count**. Derives it from the single `for _g in …; do` loop in `test/regression.sh` and rewrites all eight marked sites across `README.md`, `EVALS.md` and the deck; `--check` is the drift comparison that `test/gatecountcheck.sh` runs. Never edit that number by hand — two lanes hand-writing the same N+1 auto-merge clean against a loop of N+2. | | **[`limits_classes.tsv`](limits_classes.tsv)** | Maintainers | Cap name -> INDEXING or OUTPUT, the taxonomy `LIMITS.md` renders in its `class` column: does this cap bound what can EVER be found, or only what is shown from what was found. A sidecar with a known expiry — the tag belongs on the declaration in `src/` — kept honest by `limitstablecheck.sh`, which fails if a row names a cap that no longer exists. | | **[`lineage-paper-dates.tsv`](lineage-paper-dates.tsv)** | Maintainers | arXiv id -> publication date for every 2026 paper in `LINEAGE.md`. The ID stem does not track the date (`2607.09691` was published 2026-06-19), so the README's recency claim is re-derived from this file by `readmedriftcheck.sh` arm (H2) rather than from the ids. Adding a 2026 paper without a date row fails that arm. | +| **[`research/`](research/)** | Anyone working on an open question here, or offering to help with one | Investigation notes: a question this tool has not answered, the measurement run against it, and the pre-registration that says what would license a change. Not a claim surface — a number here is local evidence for a decision not yet taken, and nothing in it is quoted on `README.md`. | | **[`assets/`](assets/)** | The front page | The README banner and tagline artwork (SVG, self-contained). | | **[`captures/`](captures/)** | Maintainers, and the curious | One recorded run of every verb against a real repository — the source of `COMMANDS.md`'s sample output, and the harvest source for the differential argv harness. | diff --git a/docs/research/readability-construct-validity.md b/docs/research/readability-construct-validity.md new file mode 100644 index 000000000..2b57a793d --- /dev/null +++ b/docs/research/readability-construct-validity.md @@ -0,0 +1,811 @@ +# Readability construct validity: what the lens claims, and how we would check it + +Status: **investigation, not a conclusion.** This document states what `--readability` actually computes +and actually claims (read from the code, not assumed), designs a validation of its *ordering* against human +judgment that we have not yet run, and runs the two validations that need no human labels at all — reporting +their numbers whether or not they are flattering. It ends with what we would like help with from readability +researchers who study exactly this gap. + +Two published results shaped where ripwire's readability lens sits, and both are **design influences, not +implementations** — nothing in `src/readability.h` runs a model or a judge: + +- LLM judges of code readability lean on surface features rather than the construct itself — part of why + ripwire's readability lens is a deterministic, closed-form formula (Halstead volume, token entropy, the + Posnett sigmoid fit) instead of a model call, and why the underlying evidence is disclosed (`vol=`, `ent=`, + `posnett=`) rather than a single opaque number. +- A study of prompt design's association with LLM-generated code readability found the overall role of + prompt design bounded (Ye, Ran, Xu & Zhou, arXiv:2605.13280) — part of why ripwire's readability-adjacent + gate (`--quality-delta`'s `verbosity` kind) runs **after** each edit, against the actual diff, rather than + living inside a generation prompt as one more instruction competing with the rest of the prompt for + effect. + +Both are cited in the readability design note — an internal working document, not part of this +repository — at its §7, as the prompt-design-bounded finding (Ye et al., arXiv:2605.13280) and as "LLM +self-judges … fixate on surface features (CoReEval)" — see the reference list at the end of this document +for both arXiv identifiers. That design log already states +the honesty bound this document expands on: *"Everything here is a lens, never a verdict."* This document is +the follow-through on that bound — checking it empirically rather than repeating it. + +## 1. What we actually compute, and what we actually claim (from the code) + +`--readability` (`src/readability.h`) is a single closed-form pass per function or method: + +- **V — Halstead volume**: `V = N · log2(η)`, `N` = operator+operand token count, `η` = distinct tokens. +- **E — token entropy**: Shannon entropy (bits) of the definition's own token-frequency distribution. +- **L — lines**: the physical line span of the definition (`Symbol::loc`), signature included. +- **P — Posnett score**: `P = sigmoid(8.87 − 0.033·V + 0.40·L − 1.5·E)` — the coefficients published in + Posnett, Hindle & Devanbu, *A Simpler Model of Software Readability*, MSR 2011. + +Three facts about how this is used, verified against the source rather than assumed: + +**It is emitted, and used elsewhere in the tool, as an ORDERING, never as a grade.** The verb sorts rows +*least readable first* and says so in its own legend (`src/readability.h`, `kReadabilityLegend`): + +> "P was fitted on snippets of 20 lines or fewer: read the ORDER, not the number, and never as a grade." + +The header comment states the reason in more detail, and names the literature that forced this framing +rather than a house preference: *"Scalabrino ASE'17 and Trockman MSR'18 both find no readability metric +correlates strongly with measured understandability; Fakhoury ICPC'19 finds the classic models miss real +readability-improving commits. So P is never a grade, never a gate, and never a verdict — it orders a +worklist."* The published `README.md` repeats the same claim as an external-facing promise, in a table of +lenses deliberately kept *outside* the tool's evidence-weighted panel join. Its row for `--readability` +reads, across that row's "why beside the join" and "on this repo" cells: *"the fitted score saturates past 20 +lines — only the ordering is meaningful, and an ordering cannot vote in a count"* / *"ordering only, never a +grade."* + +**It is not one of `--quality-delta`'s ten gating kinds.** The per-edit gate's kinds are `complexity`, +`verbosity`, `nesting`, `params`, `duplication`, `dead-code`, `api-surface`, `error-masking`, +`short-horizon-churn` and `new-clone-of-reused-helper` (`src/quality.h:4565` and the kind dispatch around +it). None of them is Halstead volume, token entropy or the Posnett score. The one readability-*adjacent* +kind is `verbosity`, and reading its implementation shows exactly what it measures and what it does not: +`verbosity` is **CODE line count only** (`src/quality.h`, `locBySym`, `codeLocByNode` — "Q-DIAL-3: blank and +comment-only lines are not debt"). It shares one input (`L`) with the Posnett formula and none of the other +two (`V`, `E`). So the per-edit gate that actually blocks a merge never sees the entropy or Halstead-volume +half of the readability lens at all — it sees a plain, un-weighted line count. + +**The only place the Posnett rank feeds a joined judgment is `--ensemble`, and even there it is kept +ordinal, not additive.** `src/ensemble.h` folds `--readability`'s rank into its "structural" evidence family +alongside complexity/LOC/nesting/params, and its own header comment states why the five are one family and +not five independent votes: *"ccx, loc, nest, params and the Posnett score all track SIZE — Posnett's own fit +is literally linear in L. Counting them as five agreeing witnesses is the Maintainability-Index failure +(§3.10): re-weighting one signal and calling it five."* — and separately, on why the Posnett rank and churn +are handled differently from the other four: *"The other two signals (Posnett readability, churn) are +RANKINGS whose own authors publish no defensible absolute cut: --readability's header says in so many words +to read the ORDER, not the number. The only honest predicate on an ordinal signal is an ordinal cut, so each +fires for the WORST DECILE of its own ranking."* `--ensemble` records only a symbol's position in that +worst-decile prefix (`rrank=`), never `posnett=` itself — there is, by the same comment's own words a few +lines later, "no composite number anywhere in this verb, by contract." + +**So: does the tool present readability as an absolute grade anywhere?** No surface we found does. The one +place a reader could plausibly *read* it that way — a raw `posnett=` value on an individual row of +`--readability`'s own output — carries the legend's own correction directly above it every time it is +printed ("read the ORDER, not the number, and never as a grade"), and the value visibly saturates to +`0.000` for a large share of real functions (see §3), which is itself a standing, unavoidable reminder that +the number is not meant to be read on its own. We consider the code's claim to already be the *honest* one: +ordering-only, disclosed as ordering-only, at every point of contact. The rest of this document is about +whether that ordering claim itself survives contact with a human judgment — which is a strictly harder bar +than "is the code's rhetoric honest," and one the code freely admits it has not cleared (Scalabrino/Trockman/ +Fakhoury are cited as open problems, not as solved ones). + +## 2. A validation of the ORDERING, not the score + +Reproducibility (the same input always yields the same rank) is not construct validity (the rank means what +we say it means). The formula could be perfectly deterministic and still rank functions in an order no human +reader would recognize. Here is the validation we think would settle that, specified concretely enough to run: + +**Unit of comparison: pairs, not scores.** Ask a rater "which of these two is more readable," never "rate +this snippet 1–5" — the literature's own ground-truth datasets are pairwise-derived or scale-based and Vitale +(2025, cited in the readability design note §7) found up to a third of classic scale labels +self-contradictory on reread. A forced pairwise choice is cheaper to collect, cheaper to get consistent, and +is exactly the shape `--readability`'s own claim needs checking against: it emits an order, so validate the +order. + +**Where pairs come from (three tiers, cheapest first):** + +1. **This repo's own git history** — a commit whose message says refactor/simplify/cleanup gives a free + (before, after) pair of the same function, no synthetic construction needed. This is what §3's proxy (a) + already mines, at zero marginal data-collection cost. +2. **`bench/external/arb` and `bench/external/swex`** — this repo already vendors multiple external-repo + snapshots and commit histories for other evaluation harnesses (agentic-benchmark corpora, not readability + ones). The same commit-message mining in §3's script generalizes to any of them by pointing `--root` at a + different checkout — more languages, more authors, no ripwire-specific style bias. We did not run this for + the current round (time-boxed to this repo, see §3) but the script takes a `--bin`/corpus-root pair for + exactly this reason. +3. **Synthetic pairs from `--readability`'s own ranking** — take two functions already far apart in the + lens's own order (one from the worst decile, one from the best) and ask a rater to confirm or reject the + implied direction. Cheapest to collect, weakest signal (it can only confirm the lens agrees with itself at + the extremes, which §3 proxy (b) already establishes for free without a human). + +**How a human judgment would be collected.** Blinded, randomized left/right presentation (no file path, no +`posnett=`, no commit message); each pair rated independently by at least three raters, majority vote as the +pair's label, inter-rater agreement reported alongside the correlation (a low agreement number is itself a +finding, per Vitale). Raters should not be told which side the lens preferred — Piantadosi et al.'s "readable +state flip" framing (cited in the readability design note §6) is a reasonable model for how to phrase the +question without anchoring the rater on a metric. + +**The statistic.** Percentage pairwise agreement between the lens's implied direction and the majority human +label, plus a rank correlation (Kendall's τ or Spearman's ρ) over any batch of pairs drawn from one shared +ranked list, since τ is exactly a transform of pairwise agreement and gives readers a standard number to +compare against the published literature's own correlations for competing metrics. + +**What would count as success, and what would make us withdraw or demote the lens.** We propose three bands, +calibrated against the literature already cited in this repo (Scalabrino ASE'17 and Trockman MSR'18 found +*no* classic metric correlates strongly with measured understandability — so we are not calibrating against +"strong correlation is achievable," we are calibrating against "is this metric earning the narrow claim it +actually makes"): + +- **τ (or equivalent pairwise agreement) comfortably above chance and stable across a second, disjoint + sample** → keep exactly as-is: an ordering signal, disclosed as such, feeding `--ensemble`'s rank-only + join and nothing stronger. +- **Weak but directionally consistent, or consistent on some function shapes and not others (e.g., holds for + size-dominated differences, fails when two functions are close in size but differ in naming/structure)** → + narrow the claim further and say so in the legend — e.g. disclose that the lens is known to track length + more than it tracks anything len-independent, which §3's proxy (b) already shows structurally (see the + "what this cannot catch" paragraph there). +- **No better than chance, or a sign flip on a held-out language/corpus** → withdraw the ranking claim from + any joined surface (`--ensemble`'s `rrank=`) and keep `--readability` only as a standalone, clearly-labeled + "how does this formula see the codebase" report — the same demotion path `naminglens.h` already used once + (§5). + +We have not run the human-rated arm. It needs raters we do not have in this pass; §6 names it as the thing we +would like the most external help with. + +## 3. What we ran WITHOUT human labels + +Two proxies need no rater at all. Both are implemented as re-runnable scripts against the real +`./build/ripwire --readability` binary — not a reimplementation of the formula — so what they measure is +the shipped lens, not a paper description of it. + +### 3a. Refactor-commit direction (`bench/readability_refactor_pairs.py`) + +For every commit in this repo's history whose subject reads as refactor/simplify/cleanup (case-insensitive, +`refactor|simplify|clean(\s|-)?up`) and whose total changed-line count is ≤400 (so a per-function delta stays +attributable to the named refactor rather than to an unrelated bulk edit riding along in the same commit), +the script extracts every touched `.h`/`.cpp` file at the commit and at its parent, scores each version with +the real lens in a single-file scratch directory, and keeps the functions present in both versions with a +**different** token shape (same shape ⇒ this diff did not touch that function's body ⇒ no evidence either +way). Comparison uses the pre-sigmoid `z` score recomputed at full precision from the CLI's own +integer `toks=`/`vocab=` and its `ent=`/`lines=` — not the CLI's own 3-decimal `posnett=` attribute, which +saturates to a repeated `"0.000"` for most real functions (see the tie count below) and would make most pairs +falsely indistinguishable. `z` is monotonic in `posnett` (same sigmoid), so ranking by `z` ranks identically +to ranking by `posnett` without the display truncation. + +**Run** (`--max-commits 80 --max-changed-lines 400`, history pinned to the `v0.6.2` tag — never `git log +--all`, the numbers below): + +| | | +|---|---| +| candidate refactor/simplify/cleanup commits | 80 (77 contributed ≥1 usable pair) | +| function pairs (changed shape, matched by name) | 412 | +| lens ranks AFTER more readable (`z` increased) | 154 (37.4% of the 412 directional pairs) | +| lens ranks AFTER less readable (`z` decreased) | 258 (62.6%) | +| exact tie | 0 | +| mean Δz (after − before, + = more readable) | **+3.79** | +| median Δz | **−0.56** | +| pairs with \|Δz\| > 20 (large swings) | 32 / 412 (7.8%) | + +First recorded as 484 pairs / 146 right-direction / 30.2% from an unpinned `git log --all` walk, which does +not reproduce; see §3c for the instrument defect, the fix, and the full decomposition. + +**Read this plainly, including that it is not flattering.** On a majority (63%) of the function pairs a human +called a refactor, the lens's own ranking moved the *wrong* direction — it scored the after-version as *less* +readable. The mean is positive only because a small number of large swings (7.8% of pairs, all from commits +that genuinely shrank a function by splitting work into named helpers) pull it there; the median, which a +skewed distribution like this one should be read against, is negative. This lines up with exactly the +literature the lens's own code already cites as a reason for caution — Fakhoury ICPC'19's finding that +classic readability models "miss real-world readability-improvement commits" is not a hypothetical risk here, +it is what this run measured on ripwire's own history. + +**Two worked examples, to show what is and is not driving the number.** The eight largest positive swings on +the pinned population are all commits whose stated purpose was extraction — moving a block of logic out into +named helper functions, which mechanically shortens (or hollows out) the function the lens is scoring: +`buildScipOverlay` (227→124 lines), `ingest` (1138→1029), `writePinCensus` (88→38), `buildFieldNarrowTables` +(114→50), `computeSnapshot` (116→61), `buildScopedRecvDecls` (61→15), `hasNetExfilShape` (59→17) and +`buildExternalVetoTables` (147→86) — a 41%–75% line-count cut in all but one case. The exception is `ingest` +(`refactor(ingest): the doc post-pass becomes ingest_docpass.h`): only a 9.6% line cut, but the swing is +still large because a token-dense block moved out into the new header — the Halstead-volume term collapsed +even though the line count barely moved. That is the same volume-not-lines mechanism §3c's decomposition +finds behind 91%+ of the wrong-direction pairs too (see below), not a second, unrelated effect. The single +largest *negative* swing is the opposite shape of edit: commit `15af398e` +(`refactor(L1-fix): keep existing contracts; readings spell no element markup`) inlined a one-line forwarding +wrapper (`classify()`, 5 lines) into what had been a same-named sibling implementation, producing one 126-line +function where there had been a 5-line indirection plus a separate body. The commit message is about +preserving an API contract, not about readability, and a human reader might reasonably call the *pre*-commit +two-function shape less readable (an unexplained one-line forwarder) than the *post*-commit single function — +the opposite of what the lens's length term rewards. We are not correcting for this by hand; it is exactly +the kind of case a pairwise human study (§2) would need to arbitrate, and we would be misrepresenting the +proxy if we quietly excluded it. + +Re-run: `bench/readability_refactor_pairs.py --max-commits 80 --out pairs.tsv` (defaults to this repo, +`build/ripwire`, and the `v0.6.2` ref — override with `--ref`, never `--all`; deterministic given a fixed +git history and a fixed binary). + +### 3b. Self-consistency under meaning-preserving rewrites (`bench/readability_self_consistency.py`) + +Two mechanical rewrites that should not change how readable a function is: renaming one local identifier to a +fresh name that collides with nothing else in the function, and swapping two adjacent, mutually-independent +simple statements (`lhs = rhs;`, including one-line typed declarations, in this repo's one-statement-per-line +house style). 80 functions were sampled at even percentile positions across `--readability`'s own +least-readable-first ranking of this repo's `src/` (4,373 functions measured), so the sample spans the whole +distribution rather than only the worst decile the verb shows by default. Each sampled function's own whole +file is re-scored unmutated as the baseline (same single-file measurement path the mutant takes, so there is +no cross-file ingest difference to explain away), then rescored after one mutation. + +| mutation | attempted | exact tie (`vol=`, `ent=`, `posnett=` all unchanged) | diverged | +|---|---|---|---| +| rename (fresh, non-colliding identifier) | 75 / 80 | 75 | 0 | +| reorder (two adjacent independent statements) | 20 / 80 | 20 | 0 | + +**State plainly what this does and does not show.** Halstead volume and Shannon entropy are functions of the +*multiset* of (token, frequency) pairs — not of token identity or statement order (`readability.h`'s own +determinism note: the entropy sum runs over a token-text-**sorted** vector specifically so iteration order +cannot reach the output). So exact invariance under a non-colliding rename or an independent-statement +reorder is what the formula predicts *mathematically*, not a discovery — a 100% tie rate here mostly certifies +that the shipped implementation has no stray non-determinism or order leak, which is a real and worth-having +guarantee, but it is an implementation-correctness result, not a readability-construct-validity result. The +informative failure mode this proxy could have caught — and did not — is any pair that does NOT tie exactly, +which would point at either a bug in this harness's mutation or a genuine order/identity sensitivity in the +lens worth filing as a defect. + +**The same invariance is also a disclosed limitation, not only a safety net.** Because the formula is +mathematically blind to identifier length and to statement order, it is *structurally incapable* of +rewarding `numberOfActiveConnections` over `n`, or a well-ordered proof sketch over a scrambled one — both +of which most human readers would call a real readability difference. That is a construct-validity gap this +proxy makes visible without needing a single human label: whatever the lens is measuring, it provably is not +measuring those two things, by the formula's own definition. + +**Coverage caveat, reported rather than hidden.** Only 20 of 80 sampled functions (25%) contained an eligible +adjacent, independent, simple-statement pair by our conservative heuristic (excludes calls, subscripts, +member targets, and any pair whose right-hand sides cross-reference the other's left-hand side). This is a +coverage limit of the *harness*, not a claim about the lens — most real functions in this codebase either +have fewer than two adjacent simple statements or have dependencies between them, and a looser heuristic +risks constructing a swap that is not actually independent. + +Re-run: `bench/readability_self_consistency.py --samples 80 --seed 20260918 --out consistency.tsv` +(deterministic for a fixed seed, corpus and binary — the sampler's own RNG is seeded, and the rename target +is chosen by frequency-then-lexical order rather than randomly, so re-running with the same seed reproduces +the same mutations). + +### 3c. A second, independent proxy: commits that DECLARE a readability improvement + +§3a leaves two explanations standing, and they predict different things: + +- **Lens defect** — the formula genuinely disagrees with readability. Predicts the inversion reappears on + *any* readability ground truth, including commits whose author says in so many words that the change is + for readability. +- **Proxy defect** — a "refactor" commit is not a readability improvement (refactors add guards, split + responsibilities, keep contracts), which is also the published result this lens already cites (Fakhoury + ICPC'19: classic models miss developer-declared readability improvements). Predicts that the §3a inversion + is explained by a mechanical component of the formula responding to non-readability edits, and says nothing + in advance about declared-readability commits. + +The stop rule this section runs under: **if a second, independent proxy also comes out inverted, we withdraw +the ordering claim rather than re-fit the formula to rescue it.** + +#### Protocol (fixed in this commit, before any number below was computed) + +This protocol was committed to this document before the harness that computes it was run; the commit that +adds this subsection predates the commit that adds its results, and nothing here was edited afterwards. + +**Corpus and binary.** Exactly §3a's: this repository's own git history, walked with the same pinned `v0.6.2` +ref over `src/` C/C++ sources (originally `git log --all` — see the instrument-fix note in Results below), +the same ≤400 total-changed-lines cap, the same single-file scratch scoring through the shipped +`--readability` verb, the same (basename, name) matching, the same "changed token shape" filter and the same +full-precision pre-sigmoid `z` comparison. The binary is a v0.6.2 build; `src/readability.h` is byte-identical +between that build's source and this document's base. + +**Selection rule — proxy (c), primary arm.** A commit is selected when its **subject line**, after the lens's +own name is masked, matches the case-insensitive regex + +``` +readab|legib|clarity|clarif|clearer|easier to (read|follow|understand)|self-documenting +``` + +The mask replaces these substrings with a blank before matching, so a commit *about the lens* is not read as +a commit *declaring readability*: `--readability`, `readability.h`, `readability_`, and a conventional-commit +scope `(readability)` / `(readability,` / `readability)`. A subject that **also** matches §3a's refactor regex +(`refactor|simplify|clean(\s|-)?up`) is excluded, so the primary arm's commit set is disjoint from §3a's by +construction. No `--max-commits` cap: every qualifying commit in the history is used. + +**Secondary arm (descriptive only, no verdict).** The same rule applied to the full message (subject and +body), still masked and still disjoint from §3a. Bodies in this repository are long and mention readability in +passing, so this arm is noisier; it is reported because it was named here, and it carries no decision. + +**n target.** **100 directional function pairs** in the primary arm (at p = 0.5, 100 pairs gives a 95% +interval of about ±10 points, which is the width of the bands below). If the primary arm reaches fewer than +100, the result is reported as "n not reached", with the n that was reached and its rate shown for the record, +and **no verdict is drawn**. The regex is not loosened after looking. + +**Decision bands** (on the primary arm's fraction of directional pairs the lens ranks AFTER more readable): + +- **≥ 60%** → the lens's ordering holds on this proxy; §3a's inversion is read as a proxy defect. +- **≤ 40%** → inverted on a second, independent proxy; the stop rule triggers and this document recommends + withdrawing the ordering claim. +- **between** → inconclusive, reported as such; neither explanation is favoured by proxy (c). + +Because pairs cluster inside commits, the per-commit rate (a commit counts as "right" when most of its +directional pairs moved up) is also reported. It is secondary and does not move the band. + +**§3a decomposition (deterministic, fixed here too).** For each §3a pair, Δz splits exactly into three terms: +`ΔV·(−0.033)` (Halstead volume), `ΔE·(−1.5)` (entropy), `ΔL·(+0.40)` (lines). For every wrong-direction pair +(Δz < 0), the **driver** is the term with the most negative contribution. Reported: the driver split, the +mean contribution of each term over the wrong-direction pairs, and how many wrong-direction pairs got *longer* +(ΔL > 0). The same split over §3a's right-direction pairs is reported beside it for contrast. The pairs are +regenerated by re-running §3a's exact command (`--max-commits 80 --max-changed-lines 400`), now pinned to the +`v0.6.2` ref by default — walking `git log --all` instead would see every ref in this clone's shared `.git` +(~291 worktree branches at the time this defect was found), so a later ref set could silently shift which 80 +commits are "newest" and move the published fraction with it. That is exactly what happened to §3a's first +recorded number; see the instrument-fix note below. + +No LLM judgment is used anywhere in this section — not for labels, not for spot-checks. The reference list +cites CoReEval for why: an LLM readability judge fixates on surface features, which is the very construct +question under test. + +#### Instrument fix (this revision) — read this before the numbers below + +The first recorded run of this section (484 pairs / 146 right-direction / 30.2% in §3a; the proxy (c) and +decomposition numbers this subsection used to report) walked `git log --all` in both +`bench/readability_refactor_pairs.py` and `bench/readability_declared_pairs.py`. This clone's `.git` is +shared across every worktree in the orchestration tree that produced this document — at the time the defect +was found, that was on the order of 291 branches — so `--all` pulled in whatever those branches happened to +contain that day. A rerun of the *identical, unmodified* script gave 413 pairs (38.3%) with the commit window +pinned to the original run's date and 409 pairs (38.4%) without even that pin; neither matched the published +484/146/30.2%, and neither run had touched `src/readability.h` or the lens in any way. A number that moves +when an unrelated lane is pushed is not measuring the lens — it is measuring which branches exist in the +shared `.git` right now. That is an instrument defect, not a finding. + +Both scripts now default to walking exactly one immutable ref — the `v0.6.2` tag, which is also the ref the +scoring binary (`build/ripwire`, `built_from=15a20855c`) was built from — instead of `--all`. `--ref` +overrides it for a deliberately different population; `--until` (added when the defect was first worked +around) still narrows further within whatever ref is walked. §3a's headline table above is this pinned run. +Every number for the rest of this section is the same pinned run too, so — unlike the first version of this +document — the "published" and "regenerated" numbers here are one run, not two. + +#### Results (recomputed on the `v0.6.2`-pinned instrument) + +**Proxy (c), primary arm: n not reached — no verdict.** The pre-registered subject-line rule selected **3 +commits yielding 3 directional pairs** against a target of 100 (2 ranked more readable after, 1 less; 2/1 per +commit). Under the protocol that is reported as "n not reached" and draws no band. It is also worse than +small: **none of the three is a readability declaration.** Two match `readab` inside *unreadable* (commits +about files the tool could not read), and one matches the lens's own name in a form the mask did not +anticipate (`readability-wave1`, a lane name). The mask gap is a defect of the pre-registered rule, disclosed +here and not patched after the fact. The honest summary: **this repository's `src/` history, at `v0.6.2`, +holds no commit whose subject declares a readability improvement**, so proxy (c) cannot be run on this corpus +at all. + +**Secondary arm: descriptive only, and not a second proxy.** Subject and body together selected 46 commits +and 134 directional pairs, of which the lens ranked AFTER more readable on 38 (28.4%; 13/24/2 per commit +right/wrong/split). This arm carries no verdict by the protocol, and reading its commit list shows why it +should not: its subjects are overwhelmingly error-handling fixes (a file that "could not be read", a +directory walk that was "unreadable"), matched through words like *unreadable* in the body. It is a set of +bug-fix commits, not a readability ground truth, and a 28% figure on it says nothing about the lens-defect +explanation. + +**§3a decomposition.** Re-running §3a's command against the pinned `v0.6.2` ref reproduces §3a's own headline +exactly: 412 pairs (154 up, 258 down, 37.4% right-direction). The table below is that same population. + +| over the 412 pinned pairs | wrong-direction (258) | right-direction (154) | +|---|---|---| +| driver = Halstead volume term `ΔV·(−0.033)` | **235 (91.1%)** | 148 (96.1%) | +| driver = lines term `ΔL·(+0.40)` | 23 (8.9%) | 6 (3.9%) | +| driver = entropy term `ΔE·(−1.5)` | 0 | 0 | +| mean contribution: volume / entropy / lines | −3.07 / −0.06 / +0.18 | +19.79 / +0.20 / −4.91 | +| function gained tokens | 239 | 3 | +| function got longer in lines / shorter / same | 47 / 51 / 160 | 8 / 127 / 19 | + +Three things follow, all mechanical: + +1. **The inversion is a token-count signal, not a length signal.** In the Posnett fit the lines coefficient + is *positive* (+0.40): holding volume fixed, a longer function scores *more* readable. So line count cannot + be what pushed a pair the wrong way unless the function got shorter, and on average it pushed the other + way (+0.18). The Halstead-volume term drove 91% of the wrong-direction pairs; entropy drove none. +2. **Direction is almost entirely the sign of the token-count change.** On 388 of the 404 pairs whose token + count actually changed (96.0%), the lens's direction is predicted by whether the function lost tokens + (ranked more readable) or gained them (ranked less readable); 8 of the 412 pairs kept the same token count. +3. **The typical wrong-direction pair is a small addition.** Median over the 258: +5 tokens, 0 lines, Δz + −1.26 — a guard, a check, or an extra argument inside an unchanged line span, in a commit whose subject + said "refactor", "simplify" or "clean up". + +**Which explanation the data favours.** The lens-defect explanation is **untested**, not refuted: its +distinguishing prediction needs a declared-readability ground truth, and this corpus does not contain one. +What the data does settle is the mechanism behind §3a's number, and that mechanism is the shape the +proxy-defect explanation predicts: the lens did exactly what its formula says — it ranked a function that +grew by a few tokens as less readable — on commits that mostly grew functions by a few tokens. Whether those +small additions made the code more readable is precisely what a commit subject containing "refactor" does +not tell us. §3a is therefore better read as *"on this history, `--readability`'s direction is the sign of +the token-count change"* than as *"the lens is wrong 63% of the time"* — and the first statement is a +narrower, checkable fact about the lens that does not depend on the label at all. + +**The stop rule did not trigger.** It needs a second independent proxy to come out inverted, and proxy (c) +produced no verdict. The shipped `--readability` flag and its help text are unchanged by this section. One +narrowing is supported by the decomposition regardless of how the construct question resolves, and falls in +§2's middle band ("disclose that the lens is known to track length more than it tracks anything +len-independent"), refined by what was measured — it tracks *token count*, not lines. As a **proposal for +owner sign-off, not a change made here**, the legend could add: *"On ripwire's own history the ORDER between +two versions of a function followed the sign of its token-count change in 96% of pairs; read a move as 'more +or fewer tokens', not as more or less readable."* + +**Next step.** Proxy (c) needs a corpus that actually contains declared-readability commits — Fakhoury et +al.'s 548-commit set is the natural one — scored with this same script by pointing it at that checkout, under +this same protocol (bands, n target, and the stop rule unchanged; the mask gap above fixed *before* that run +and disclosed as a change). + +Re-run: `bench/readability_declared_pairs.py --bin build/ripwire` (proxy (c), both arms, `v0.6.2`-pinned by +default) and `bench/readability_declared_pairs.py --bin build/ripwire --decompose` (the §3a decomposition, on +the same pinned population as §3a itself). `--ref` points either command at a different immutable ref; +`--until` still narrows within whichever ref is walked. + +## 4. Does the ordering predict anything worth acting on? A later-fix-rate validation + +§2 and §3 both validate the ORDER: does it move the direction a refactor implies, is it stable under a +meaning-preserving rewrite. Neither asks the question an agent actually relies on when it reaches for +`--readability` to pick a target: **does a low score predict that a function will need fixing later?** That +is the implicit claim behind "use the worst-ranked functions as a worklist," and it has never been tested. +This section tests it, with no human label and no LLM judge, against a complexity control this repo's own +`--quality-delta` "complexity" gate kind already trusts, on the same population, with and without controlling +for function size. + +### Protocol (fixed in this commit, before the harness that computes it was run) + +This subsection was committed before `bench/readability_fixrate_validity.py` was run for the record — the +same discipline §3c used for proxy (c), and for the same reason: a band decided after seeing the number is +not a band. + +**Population.** Every `fn`/`method` in `src/**/*.{h,hpp,cpp,cc}` at a single pinned CUTOFF commit on +`v0.6.2`'s history — never `--all` (§3c's instrument-fix note explains why a shared `.git` makes that +non-reproducible). The cutoff is chosen to leave a multi-week, thousand-plus-commit follow-up window to the +pinned UNTIL ref (`v0.6.2` itself), long enough that a fix-shaped commit has real room to happen, short +enough that the population is still recognizably today's codebase. Functions are matched between a real +`--readability` crawl and a real `--metrics` crawl of the identical extracted tree (`git archive`, not +per-file scratch, so `--metrics`'s in/out/ccx see real cross-file structure) by (path, name), requiring +`loc`(metrics) `==` `lines`(readability); an ambiguous match (same path+name, disagreeing loc — almost always +an overload) is dropped and counted, never guessed at. + +**Exposure**, per function, read at the cutoff: `z`, the exact pre-sigmoid Posnett score recomputed from +integer `toks=`/`vocab=` exactly as `bench/readability_refactor_pairs.py` does (lower z = less readable — +`--readability`'s own least-readable-first sort order), and `ccx`, cognitive complexity from `--metrics` — +the same metric `src/quality.h`'s `"complexity"` gate kind reads, used here as the already-trusted baseline +on the identical population, not as a bar the lens must clear. + +**Outcome**, defined before looking, over the window `(CUTOFF, UNTIL]`: + +- **fix-shaped commit**: subject line (not body — §3c already found body-matching noisy) matches, case + insensitively: `\bfix(e[sd])?\b|\bbug(s|fix(e[sd])?)?\b|\bcrash(e[sd])?\b|\bregression(s)?\b`. +- a function is **FIXED** if any fix-shaped commit in the window has a diff hunk — in that commit's OWN + parent's line numbers, not the cutoff's — overlapping the function's span AS MEASURED AT THAT PARENT + revision (re-scored per commit specifically so line drift from earlier window commits cannot misattribute + a hunk to the wrong function), matched back to the population by (basename, function name). A pure-insertion + hunk (old count 0) is treated as touching whichever function(s) contain old-line `start` or `start+1`. +- a function is **MODIFIED** (the broader arm) under the identical rule over every commit that touches a src + file in the window, fix-shaped or not — a superset computed in the same pass, since every fix-shaped commit + is also a modifying commit. + +**Statistic.** Fix-rate (and modified-rate) in the least-readable quartile (bottom 25% by z) vs the rest, +Wilson interval per proportion, risk ratio with a Katz log-CI — reported beside the identical statistic for +the highest-ccx quartile vs the rest, on the same population, as the trusted-signal comparison point. +Repeated within three size (lines-at-cutoff) terciles, with the least-readable/highest-ccx quartile +recomputed WITHIN each tercile, to test whether either signal survives controlling for size. + +**Decision bands**, restated from this document's own §2 framing, applied to the raw (unstratified) arm: +the least-readable quartile's fix-rate CI excludes a risk ratio of 1, and the effect does not vanish once +stratified by size → keep the ordering claim as an actionable, if weak-to-moderate, signal; effect present +raw but gone in every size tercile → the size confound explains it, narrow the claim accordingly; CI includes +1 raw → withdraw the "predicts later fixes" claim outright. The complexity arm is reported for comparison, +never as a bar the lens must clear — Scalabrino/Trockman's own consensus (§2) is that no classic metric +correlates strongly with anything, so "about as good as complexity" is itself the "keep as a weak signal" +outcome, not a pass/fail line of its own. + +**A pre-registered caveat about the size control itself.** §3c already found that `z`'s direction on this +repository's own history is almost entirely the sign of the TOKEN-count change (96.0% of pairs), not the +line-count change — the Posnett lines coefficient is positive, so more lines alone would push z the other +way. Stratifying by LINES (the natural, legible size band) therefore does not fully neutralize the mechanism +§3c already implicated: two functions in the same line-count tercile can still differ sharply in token +volume, and z tracks that. A result that survives a lines-based stratification is not automatically a result +that survives a tokens-based one — that is disclosed here, before either number exists, as a limitation of +this design, not folded quietly into the verdict after the fact. + +Re-run: `bench/readability_fixrate_validity.py --bin build/ripwire` (population + both outcome arms, raw and +size-stratified, `v0.6.2`-pinned CUTOFF/UNTIL by default; `--out` writes the per-function TSV this section's +numbers were computed from). + +### Results + +**Population and window.** 2,868 of 3,004 functions crawled at the cutoff (`4f5c310c`, 2026-09-03) matched +between the `--readability` and `--metrics` passes (95.5%; 67 ambiguous matches dropped). The follow-up +window to `v0.6.2` (2026-09-21) held 1,138 commits touching a `src/` file, 556 of them fix-shaped by the +pre-registered regex (48.9%). + +**Raw (unstratified).** + +| outcome | arm | worst-quartile rate | rest rate | risk ratio (95% CI) | +|---|---|---|---|---| +| FIXED | readability (least-readable quartile, n=717) | 386/717 = 0.538 [0.502, 0.575] | 403/2151 = 0.187 [0.171, 0.204] | **2.87 [2.57, 3.21]** | +| FIXED | complexity (highest-ccx quartile, n=717) | 339/717 = 0.473 [0.437, 0.509] | 450/2151 = 0.209 [0.193, 0.227] | **2.26 [2.02, 2.53]** | +| MODIFIED | readability | 495/717 = 0.690 [0.656, 0.723] | 659/2151 = 0.306 [0.287, 0.326] | **2.25 [2.08, 2.44]** | +| MODIFIED | complexity | 452/717 = 0.630 [0.594, 0.665] | 702/2151 = 0.326 [0.307, 0.346] | **1.93 [1.78, 2.10]** | + +Both signals separate from 1 by a wide margin on this population, on both outcome definitions — and, raw, the +readability quartile's risk ratio is *larger* than the complexity quartile's on every row, not smaller. + +**Size-stratified** (terciles by lines-at-cutoff; n=956 each; quartile recomputed within each tercile): + +| tercile (lines) | outcome | readability RR (95% CI) | complexity RR (95% CI) | +|---|---|---|---| +| T1 (1–10) | FIXED | 3.30 [2.32, 4.70] | 0.80 [0.51, 1.24] | +| T2 (10–26) | FIXED | 1.27 [0.995, 1.63] | **0.55 [0.40, 0.77]** | +| T3 (27–1494) | FIXED | 1.90 [1.69, 2.14] | 1.39 [1.22, 1.58] | +| T1 (1–10) | MODIFIED | 2.39 [1.88, 3.05] | 1.08 [0.82, 1.44] | +| T2 (10–26) | MODIFIED | 1.23 [1.03, 1.47] | **0.79 [0.64, 0.98]** | +| T3 (27–1494) | MODIFIED | 1.56 [1.44, 1.69] | 1.24 [1.13, 1.37] | + +**Read this plainly.** The pre-registered "most likely outcome" — the signal vanishing once size is held +constant — did **not** happen. The readability risk ratio stays above 1, with a CI excluding 1, in five of +six stratified rows; it only touches 1 in the FIXED/T2 row (lower bound 0.995), and even there the MODIFIED/T2 +row for the same tercile clears 1 (1.03). It attenuates from the T1 (smallest-function) band — where it is +strongest, 3.30 and 2.39 — through T2, then rises again in T3. Complexity's within-band behavior is the more +surprising result here: it is flat-to-inverted in T1 and *significantly below 1* in T2 (0.55 and 0.79, both +CIs excluding 1) — meaning, within these two size bands, the highest-cognitive-complexity quartile was +fixed/modified *less* often than the rest, the opposite of the trusted baseline's raw-population direction. +Complexity only behaves as expected (RR > 1) in T3, the largest-function band. + +**What this does and does not show.** On the raw population, `--readability`'s ordering separates a fixed +population by later-fix rate at least as well as the complexity control does, and the effect survives a +lines-based size stratification better than complexity's own effect does. That is a real, actionable-looking +signal on this corpus, under this outcome definition — the decision band above calls this a "keep," not a +"withdraw." But the pre-registered caveat means this is not the full size-confound test: `z` is dominated by +token volume, not lines (§3c), so a function can sit in the smallest LINE tercile while carrying a large +TOKEN volume relative to its tercile-mates, and the lines-based stratification cannot separate "z predicts +fixes independent of size" from "z still partly tracks a size axis lines does not capture." A token-count +stratification (bucketing by `toks=` instead of `lines=`) is the natural next test and was not run in this +round. Separately: complexity's inversion in T2 is itself worth a second look before this repository leans on +"highest complexity" as a worklist filter for medium-sized functions — that result was not anticipated by +this protocol and is reported exactly as measured, not smoothed over because it complicates the expected +story. + +### Correction: the line-based stratification did not hold the confound constant + +The paragraph above named the gap and deferred the fix; this section runs it. The lines-based stratification +just reported does not hold the actual confound constant: §3c already established that `z`'s direction +tracks the SIGN of the token-count change on 96.0% of this repository's own refactor pairs, not the +line-count change — the median wrong-direction pair there was +5 tokens, 0 lines. A function can gain a +meaningful share of a size band's token volume with no line-count change at all, so two functions in the +same LINE tercile can differ sharply in the unit `z` actually consumes. "Keep the ordering claim" was +therefore not yet earned by the line-stratified result alone. This re-runs the identical stratified analysis +by `toks=` — the Halstead N (operator+operand count) `--readability` itself emits, the exact integer the +lens's own volume term is computed from, not a re-derived proxy — on the same population, same outcome +definitions, same CIs, using the updated `bench/readability_fixrate_validity.py` (now captures `toks=` per +function and stratifies by either measure). + +**Token-count stratification** (terciles by `toks`-at-cutoff; n=956 each; quartile recomputed within each +tercile), reported beside the line-stratified table rather than replacing it: + +| tercile (toks) | outcome | readability RR (95% CI) | complexity RR (95% CI) | +|---|---|---|---| +| T1 (10–72) | FIXED | 2.28 [1.56, 3.33] | 1.01 [0.65, 1.57] | +| T2 (73–191) | FIXED | 1.06 [0.82, 1.36] | **0.62 [0.46, 0.84]** | +| T3 (191–8023) | FIXED | 1.89 [1.68, 2.13] | 1.38 [1.21, 1.58] | +| T1 (10–72) | MODIFIED | 1.74 [1.33, 2.28] | 1.02 [0.75, 1.39] | +| T2 (73–191) | MODIFIED | 1.05 [0.88, 1.26] | 0.85 [0.70, 1.04] | +| T3 (191–8023) | MODIFIED | 1.59 [1.46, 1.72] | 1.27 [1.15, 1.39] | + +**Do the two stratifications agree?** No, and the disagreement is itself the finding. Line-stratified, +readability's RR excluded 1 in five of six rows (only FIXED/T2 touched 1, at a lower bound of 0.995). +Token-stratified — the measure the lens actually consumes — readability's RR excludes 1 in only **four** of +six rows: it holds in T1 and T3 on both outcomes, but **both** T2 rows now include 1 (FIXED 1.06 [0.82, +1.36]; MODIFIED 1.05 [0.88, 1.26]) — indistinguishable from no effect in the middle third of the token +distribution, where the line-based version had reported a real (if borderline) signal. The line-based +analysis was, exactly as suspected, partly reading a token-volume effect that a line-count band does not +hold constant. + +**Restated verdict against the pre-registered bands, using the token-stratified result as decisive.** The raw +(unstratified) separation stands unchanged (RR 2.87 [2.57, 3.21] fixed, 2.25 [2.08, 2.44] modified — both +exceed the complexity control's 2.26 / 1.93 on the same population) and by itself would satisfy the "keep" +band. But the token-stratified arm shows that separation is not uniform across the size range the lens +itself measures: it is real and CI-excludes-1 at both ends (small-token and large-token thirds) and +statistically silent in the middle third. That is neither this document's "keep exactly as-is" band (which +requires the effect not to vanish under stratification, full stop) nor its "withdraw" band (which requires +it to vanish in *every* tercile — it does not: T1 and T3 both hold). It matches the **middle band this +document's own §2 already defined for exactly this shape of result**: "consistent on some function shapes +and not others… disclose that the lens is known to track [size] more than it tracks anything +[size]-independent, refined by what was measured." The verdict is therefore: **keep the ordering claim, but +narrowed** — `--readability`'s later-fix signal on this corpus is not a uniform property of the ranking, it +is concentrated at the size extremes (very small and very large functions, measured in tokens) and +disappears for functions of middling token volume, where the raw quartile RR reported earlier is optimistic. +An agent reading "least readable" as a worklist filter should expect the signal to be weakest for +run-of-the-mill mid-sized functions — which is most of the codebase (T2 by construction) — and strongest at +the tails. This is a materially weaker claim than "keep exactly as-is," and this document says so plainly +rather than defaulting to the friendlier reading. + +**T2 diagnosis: is the complexity inversion explained by a file or symbol-kind concentration?** Complexity's +below-1 risk ratio in the middle tercile is the more surprising result in both stratifications (line-based: +0.55 fixed / 0.79 modified; token-based: 0.62 fixed / 0.85 modified, the modified arm no longer excluding 1 +under the token version). Checked cheaply, on both the line-T2 and token-T2 highest-ccx quartiles (n=239 +each): no file supplies more than 5% of either group (top file `src/docdrift.h` at 12–15 of 239), and no +name-pattern keyword (`test`, `match`, `find`, `build`, `table`, `gen`, `classify`, `parse`, `check`, …) +covers more than 3.3% — nothing resembling generated code, a table literal, or a test-fixture cluster +dominates either quartile. The sampled names instead look like a broad mix of short, branch-dense helpers +(classifiers, matchers, small validators: `walkAggregateBody`, `classifySkipHealth`, +`sliceBuildAnchorOccs`) spread across ~40 different files with no concentration. **This is reported as +unexplained.** A plausible but unverified guess — that dense, short decision functions get written once, +exhaustively branch-tested, and rarely revisited — is not checked here and is not asserted as a finding; per +this document's own rule, an anomaly without a cheap explanation is left as measured, not smoothed into a +story. + +**External validity.** Every number in this section, both stratifications and the raw arm, comes from one +repository's own history: `ripwire`'s own `src/` under this project's ~662 CI gates, quality-delta review and +adversarial CI, much of it AI-authored under those gates. A later-fix rate measured here is a claim about +*this corpus under this development process*, not a general claim about code, or even about AI-authored code +generally — a codebase without this repository's gate density, review discipline, or authorship mix could see +a different relationship (or none) between `z` and later-fix rate. Nothing in this section should be read as +"readability predicts fixes" as a general software-engineering claim; it is "readability predicted fixes on +this codebase, in this window, measured this way" — exactly the same scope limit §3's proxies already carry. + +Re-run (updated to also emit the token-stratified table and the per-function `toks=` column in `--out`): +`bench/readability_fixrate_validity.py --bin build/ripwire`. + +### Second correction (POST-HOC, not pre-registered): the tercile itself does not hold size constant + +**This subsection was prompted by looking at the token-stratified table above**, not written before it — +labelled as such, per this document's own rule about what counts as pre-registration. The tercile result +has a simpler reading than "keep, narrowed": T1 (10–72 tokens) spans a 7.2× internal range, T2 (73–191) spans +2.6×, T3 (191–8023) spans 42×. The two terciles that show a readability signal are the two with the widest +internal spread; the one that is silent is by far the narrowest. A tercile — even a token tercile — does not +hold size constant, so "the effect survives at both extremes" is exactly what a *residual* size effect +predicts too, not only what an independent readability effect would predict. The two hypotheses make +different, testable predictions on narrower bands: a genuine readability effect should not shrink as the +band narrows; a residual-size effect should shrink toward RR=1, at a rate that tracks how much internal +range is left in the band. + +**Deciles by `toks`-at-cutoff** (same population, same outcome definitions, same CIs; quartile recomputed +within each decile; `bench/readability_fixrate_validity.py --load-tsv pop_outcomes.tsv` re-derives this table +from the already-computed per-function data in under a second — no re-crawl needed): + +| decile | toks range | internal range | n | FIXED RR (95% CI) | MODIFIED RR (95% CI) | +|---|---|---|---|---|---| +| D01 | 10–29 | 2.90× | 287 | 1.12 [0.31, 4.11] | 0.66 [0.23, 1.90] | +| D02 | 29–47 | 1.62× | 287 | 1.49 [0.70, 3.18] | 1.27 [0.77, 2.09] | +| D03 | 47–66 | 1.40× | 286 | 1.86 [1.03, 3.34] | 1.15 [0.73, 1.82] | +| D04 | 66–89 | 1.35× | 287 | 1.00 [0.58, 1.71] | 1.00 [0.67, 1.47] | +| D05 | 89–118 | 1.33× | 287 | 1.36 [0.88, 2.12] | 1.19 [0.84, 1.69] | +| D06 | 118–157 | 1.33× | 287 | 0.81 [0.51, 1.28] | 0.78 [0.55, 1.09] | +| D07 | 157–211 | 1.34× | 287 | 1.38 [0.92, 2.06] | 1.16 [0.88, 1.53] | +| D08 | 212–300 | 1.42× | 286 | 0.92 [0.63, 1.33] | 0.96 [0.73, 1.26] | +| D09 | 301–491 | 1.63× | 287 | **1.72 [1.37, 2.16]** | **1.36 [1.16, 1.60]** | +| D10 | 492–8023 | **16.31×** | 287 | **1.47 [1.29, 1.68]** | **1.25 [1.15, 1.36]** | + +**Does RR track the internal range?** Yes, as the dominant pattern. Eight of the ten deciles (D01–D08, each a +genuinely narrow 1.3×–2.9× band) show a CI that includes 1 on both outcomes, with point estimates scattered +on both sides of 1 (0.81 to 1.86) — the shape of noise around no effect, not a consistent direction. The only +two deciles whose CI excludes 1 on both outcomes are D09 and D10. D10 is not a narrow band at all: at 16.31× +internal range it is wider than every tercile ever examined in this document, so a significant RR there is +exactly what "the tercile does not hold size constant" predicts, one level down. D09 (1.63×) is narrower and +its significance does not fit the range story as cleanly — at n=287 per decile with ten bands tested on two +outcomes (20 tests), roughly one false positive at α=0.05 is expected by chance alone even under a true null, +so D09 is reported without a confident causal reading in either direction, not folded into the "it's just +size" story to make the pattern look cleaner than it is. + +**A second, independent check: nearest-token-neighbour matching.** For each of the 717 least-readable-quartile +functions, its closest-toks match from the remaining 2,151 was found (§4's protocol, `nearest_token_match`). +**This check came back unusable, and is reported as such rather than as evidence either way.** Only 46 +distinct functions serve as matches for all 717 quartile members: one function (`parseDeclarator`, 327 toks, +itself FIXED) supplies 403 of the 717 matches (56.2%); a second (`printUsage`, 1,417 toks, also FIXED) +supplies another 126 (17.6%) — together 73.8% of the "matched control" group is two always-fixed functions +repeated hundreds of times, and the mean token gap between a quartile function and its nearest match is +136 +tokens (median 37, max 6,606) — not a close match at all for most of the quartile. The raw number this +produces (matched-control fixed-rate 0.868 vs the quartile's own 0.538, RR 0.62) is **not reported as a +finding**: it is an artifact of how sparse the "rest" population is near the token counts the least-readable +quartile actually occupies, not a size-controlled comparison. That sparsity is itself informative in a +different way — it confirms the quartile sits in a part of the token distribution the "rest" of the +population barely reaches, which is what "the quartile is largely a size cut" predicts — but the RR number +itself is discarded, not used. + +**Restated verdict, using the finest stratification with usable n as decisive.** The pre-registered bands +(§4's Protocol) ask whether the effect vanishes once stratified by size. On deciles narrow enough to +plausibly hold `toks` roughly constant (D01–D08), it does: eight independent CIs on each outcome, none +excluding 1, point estimates with no consistent direction. The two bands that still show a significant +effect are exactly the ones that still carry a wide internal size range (D10 unambiguously; D09 arguably, and +flagged as not cleanly resolved). That is not "keep, narrowed" — this document's second-tier band required +the effect to hold "for size-dominated differences" specifically and to be understood as tracking length; it +is closer to, and is called, this document's **withdraw** band: **the raw and tercile-level separation this +section reported earlier is better explained as a residual size effect, measured in the lens's own units, +than as an independent later-fix signal.** The "keep, narrowed" verdict in the prior revision of this section +is superseded by this one. `--readability`'s ordering does not appear to predict later fixes beyond what its +own dominant token-count component already predicts on its own — which is the same mechanism §3c already +found driving the lens's direction on refactor pairs. Per this document's own precedent (§5, `naminglens.h`), +the correct response to a validation that comes back this way is to say so plainly rather than re-fit the +analysis until it reads more favourably, which is what this subsection does: it withdraws its own prior +verdict in the same document that made it, three commits later, rather than quietly. + +**What would change this again.** A genuine readability-independent signal, if one exists, would need to +show up as a CI excluding 1 on a majority of narrow, size-matched bands — not on the single widest band, and +not on a matching check that turned out to be degenerate. A larger population (a bigger corpus, or a longer +follow-up window) would shrink the decile CIs enough to tell D09 apart from noise, and a working size-matched +control (this repo's own "rest" population is too sparse near the quartile's own token range for 1-NN +matching; a synthetic or cross-repo control pool would not have that gap) would settle the discarded check +above properly instead of leaving it unresolved. + +Re-run: `bench/readability_fixrate_validity.py --bin build/ripwire` (full run, includes deciles and the +matched-control check) or, on an already-written `--out` TSV, `bench/readability_fixrate_validity.py +--load-tsv pop_outcomes.tsv` (re-stratifies in under a second, no re-crawl). + +## 5. Precedent: we have already withdrawn a lens that failed exactly this kind of check + +This is not the first deterministic proxy ripwire has shipped, measured, and had to reckon with. §9.0 of +`docs/LINEAGE.md`, and the top of `src/naminglens.h` itself, record `naming-body-mismatch` — a rule that +flagged a name whose tokens shared zero vocabulary with its own body. Measured on this repository's own +`src/` at the commit that shipped it, the rule produced 159 of the lens's 217 naming findings (73% of the +whole signal), and the flagged set was **dominated by the best-named functions in the tree** +(`didYouMean`, `transitiveCallers`, `symbolAdjacency`) — because a good abstraction name states *intent* +while its body states *mechanism*, so near-zero overlap is the signature of a successful abstraction at least +as often as of a lying name. The axis was non-monotonic with quality: no threshold on it has a defensible +direction. It was **withdrawn before it shipped**, and `naminglens.h` carries a do-not-re-add note plus the +measured numbers at the top of the file rather than a silent deletion. + +We take that as the template for what "failing this validation" would require of us, not just for naming: if +§2's human-rated pairwise study comes back at or below chance, or flips sign across a second corpus, the +correct response is the same one `naming-body-mismatch` got — record the numbers and the reasoning where the +next reader will actually see them (this document and `--readability`'s own header comment, the way +`naminglens.h`'s header comment carries its own withdrawal), demote or remove the claim from any joined +surface (`--ensemble`'s `rrank=` first), and do **not** quietly re-add a close cousin of it later without +citing why this round's finding no longer applies. §3's numbers are not at that bar yet — proxy (a)'s +63%-wrong-direction result on refactor commits is concerning enough that we think the human-rated study in +§2 is now the right next step, not an optional nice-to-have. + +## 6. What we would like help with + +We are not readability researchers; we are reporting what a deterministic, disclosed, ordering-only lens +measures against a construct it was never claimed to solve, and we would like informed pushback on the +following, specifically: + +1. **Is §2's proposed pairwise + rank-correlation protocol the right design**, or is there a better-controlled + study shape for validating an *ordering* claim (as opposed to the score-based designs most classic + readability datasets — Buse & Weimer, Scalabrino, Dorn — were built for)? We deliberately avoided asking + for a 1–5 scale per snippet, on Vitale's (2025) finding that a meaningful fraction of such labels are + self-contradictory on reread; is a forced pairwise choice actually more reliable, or does it just move the + inconsistency somewhere this document has not thought to look? +2. **What is a defensible pass/fail bar for an ordering-only metric**, given that the field's own consensus + (Scalabrino ASE'17, Trockman MSR'18) is that no classic metric correlates *strongly* with measured + understandability? §2 proposes three bands calibrated against "does the lens earn the narrow claim it + makes," not against "strong correlation is achievable" — is that the right frame, or does it let a weak + metric off too easily? +3. **Proxy (a)'s 63%-wrong-direction number** (§3a) is the most actionable finding in this document, and we + would like a sanity check on the method before we act on it: is commit-message mining (refactor/simplify/ + cleanup) too noisy a readability label on its own — conflating "the author changed something for reasons + unrelated to readability" with "the author made it more readable" — and if so, what filter (a stricter + subject regex, a manual pass over a sample, restricting to commits that touch exactly one function) would + make the label trustworthy enough to report a headline number from? +4. **Is Halstead volume, entropy and length the right feature set to be checking at all** in 2026, or is this + entire investigation validating a formula the field has already moved past? The readability design note + §5 surveys later models (Buse & Weimer 2010, Scalabrino 2018) that add lexical/visual/textual + features on top of the same structural core — if there is a more recent, still-deterministic (no model + call) formula with better-established construct validity, we would rather adopt it than keep defending + Posnett 2011 out of inertia. + +## Reference list + +- Posnett, D., Hindle, A. & Devanbu, P. *A Simpler Model of Software Readability.* MSR 2011. + [doi:10.1145/1985441.1985454](https://doi.org/10.1145/1985441.1985454) +- Halstead, M. H. *Elements of Software Science.* Elsevier, 1977. +- Scalabrino, S., Linares-Vásquez, M., Poshyvanyk, D. & Oliveto, R. *Improving Code Readability Models with + Textual Features.* ICPC 2016 / *A Comprehensive Model for Code Readability.* JSEP 2018. + [doi:10.1109/ICPC.2016.7503707](https://doi.org/10.1109/ICPC.2016.7503707) +- Trockman, A. et al. — the MSR 2018 result cited throughout this repo's readability code as finding no + readability metric or combination correlates strongly with measured understandability (see + `src/readability.h`, the readability design note §0). +- Fakhoury, S. et al. *Improving Source Code Readability: Theory and Practice.* ICPC 2019 — 548 + developer-declared readability-improving commits across 63 projects; classic models "fail to capture + readability improvements." +- Peitek, N., Apel, S., Parnin, C., Brechmann, A. & Siegmund, J. *Program Comprehension and Code Complexity + Metrics: An fMRI Study.* ICSE 2021. [doi:10.1109/ICSE43902.2021.00056](https://doi.org/10.1109/ICSE43902.2021.00056) — + Halstead volume specifically tracks measured cognitive load. +- Vitale, T. et al. (2025) — cited in the readability design note §0 as finding up to a third of classic + readability ground-truth labels self-contradictory. +- Ye, H., Ran, F., Xu, W. & Zhou, M. *Characterizing Readability Issue Patterns and the Role of Prompt + Design in LLM-Generated Code.* arXiv:2605.13280 — prompt design's role in generated-code readability is + bounded; cited in the readability design note §7. +- CoReEval, arXiv:2510.16579 — LLM self-judges of code readability fixate on surface features; cited + alongside the prompt-constraint study in the readability design note §7 as the joint reason + `--quality-delta`'s gate is deterministic and external to the model, applied to the diff. +- `docs/LINEAGE.md` §9.0 and `src/naminglens.h` (top-of-file comment) — the withdrawn `naming-body-mismatch` + rule, this repo's only other instance of "measured, then withdrawn," and the template §5 of this document + follows.