Files
stack/packages/mosaic/framework/tools/wake/validate-973/convert-973.py
T
Jason WoltjeandClaude Fable 5 10a7fe43c5 wake/#973: convert all 261 grep-verdict sites to has_match/count_lines
Mechanical conversion driven by the frozen denominator artifact
(validate-973/convert-973.py, committed for auditability): every row is
byte-verified against the worktree BEFORE any edit — a conversion applied to
a line the denominator did not measure would be the umbrella defect wearing
the converter's clothes — then 233 sites become has_match and 28 grep -c
sites become count_lines, and each suite gains three header lines (source
_wake-common.sh + wake_assert_init) directly after its SCRIPT_DIR line, a
uniform +3 shift for every site coordinate.

Env prefixes are kept (LC_ALL=C has_match — the prefix reaches the grep
child, microtest C8). The two multi-grep pipeline sites
(digest-quarantine:560, install:420) convert BOTH greps: each element is
measurement-bearing, and under pipefail an rc=2 on the left is masked by an
rc=1 on the right — the same defect one pipe deeper. Call-site 2>/dev/null
redirects (the preimage canaries) are kept: ordinary grep noise stays
suppressed, while an abort writes to the wake_assert_init-saved real stderr.

Diff arithmetic: +291/-261 = 261 one-for-one line replacements + 30 header
lines (10 suites x 3). Nothing in this commit is not the conversion.

All ten suites run green after conversion. Per the review pre-registration
(A9) that green is NON-EVIDENCE for this change: a suite that was green
before conversion and green after witnesses nothing about whether the sites
were converted — the counted-ledger equality in the validation harness is
the evidence.

Written-by: pepper (sb-it-1-dt)
Co-Authored-By: Claude Fable 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_01NsKce8iZuSuRnu3gVMCBKB
2026-07-31 01:58:26 -05:00

132 lines
5.1 KiB
Python

#!/usr/bin/env python3
"""convert-973.py — mechanical #973 site conversion, driven by the frozen
denominator artifact (denominator-089615f.json), never by ad-hoc grepping.
For every row in the artifact it FIRST verifies the worktree line still says
what the artifact froze (exact match after whitespace strip, artifact-side
truncation as prefix match, or the normalized suite-summary template), and
aborts before touching anything on the first verification failure — a
conversion applied to a line the denominator did not measure would be the
umbrella defect wearing the converter's clothes.
Transforms (behaviour-preserving; verdict semantics unchanged on grep rc 0/1):
E-count-capture `grep -c ARGS` -> `count_lines ARGS` (helper adds -c)
all other forms `grep ARGS` -> `has_match ARGS` (drop-in)
env prefixes (`LC_ALL=C grep`) are kept — the prefix reaches the grep child
through the function (microtest C8).
The two multi-grep pipeline sites (digest-quarantine:560, install:420) convert
BOTH greps: each is measurement-bearing, and under pipefail an rc=2 in the
left element is masked by an rc=1 in the right — the same defect one pipe
deeper. They stay ONE denominator site each (one coordinate); the ledger
records helper calls, so those coordinates appear twice per execution and the
equality check compares SETS of coordinates.
Finally each suite gains three header lines directly after its SCRIPT_DIR
assignment (source + wake_assert_init + comment), shifting every site by +3
lines exactly; the validation harness maps base coordinates accordingly.
"""
import json
import re
import sys
from pathlib import Path
HERE = Path(__file__).resolve().parent
WAKE = HERE.parent
ART = HERE / "denominator-089615f.json"
RX_GREP_TOKEN = re.compile(r"(^|[^A-Za-z0-9_.-])grep([^A-Za-z0-9_.-]|$)")
HEADER = [
"# #973: three-valued grep assertion helpers (has_match/count_lines); init saves real stderr for abort loudness.\n",
"# shellcheck disable=SC1091\n",
'. "$SCRIPT_DIR/_wake-common.sh" && wake_assert_init\n',
]
MULTI_GREP_CONVERT_BOTH = {
("test-wake-digest-quarantine.sh", 560),
("test-wake-install.sh", 420),
}
SUMMARY_TEMPLATE_MARK = '$(grep -c . "$FAILFILE")'
def verify(row, actual):
a = actual.strip()
t = row["text"].strip()
if a == t:
return True
if t and a.startswith(t): # artifact-side truncation
return True
# normalized suite-summary template rows
if t.startswith('echo "wake ') and SUMMARY_TEMPLATE_MARK in actual and a.startswith('echo "wake '):
return True
return False
def convert_line(row, line):
key = (row["file"], row["line"])
n_grep = len(RX_GREP_TOKEN.findall(line))
if key in MULTI_GREP_CONVERT_BOTH:
assert n_grep == 2, f"{key}: expected 2 grep tokens, found {n_grep}"
else:
assert n_grep == 1, f"{key}: expected 1 grep token, found {n_grep}: {line!r}"
if row["form"].startswith("E"):
assert line.count("grep -c ") == 1, f"{key}: E row without single 'grep -c ': {line!r}"
return line.replace("grep -c ", "count_lines ", 1)
def repl(m):
return m.group(1) + "has_match" + m.group(2)
count = 2 if key in MULTI_GREP_CONVERT_BOTH else 1
return RX_GREP_TOKEN.sub(repl, line, count=count)
def main():
art = json.loads(ART.read_text())
rows = art["rows"]
by_file = {}
for r in rows:
by_file.setdefault(r["file"], []).append(r)
# pass 1: verify every row before touching any file
bad = 0
texts = {}
for f, frs in by_file.items():
lines = (WAKE / f).read_text().split("\n")
texts[f] = lines
for r in frs:
if not verify(r, lines[r["line"] - 1]):
bad += 1
print(f"VERIFY-FAIL {f}:{r['line']}\n artifact: {r['text']!r}\n worktree: {lines[r['line'] - 1]!r}")
if bad:
sys.exit(f"ABORT: {bad} row(s) failed verification; nothing was modified.")
# pass 2: convert + insert header
total = {"has_match": 0, "count_lines": 0}
for f, frs in sorted(by_file.items()):
lines = texts[f]
for r in frs:
i = r["line"] - 1
new = convert_line(r, lines[i])
assert new != lines[i], f"{f}:{r['line']}: no-op conversion"
lines[i] = new
total["count_lines" if r["form"].startswith("E") else "has_match"] += 1
# header insertion after the SCRIPT_DIR= line
sd = [i for i, ln in enumerate(lines) if ln.startswith('SCRIPT_DIR="$(')]
assert len(sd) == 1, f"{f}: expected exactly one SCRIPT_DIR line, found {len(sd)}"
first_site = min(r["line"] for r in frs) - 1
assert sd[0] < first_site, f"{f}: SCRIPT_DIR line {sd[0] + 1} not before first site {first_site + 1}"
lines[sd[0] + 1 : sd[0] + 1] = [h.rstrip("\n") for h in HEADER]
(WAKE / f).write_text("\n".join(lines))
print(f"{f}: {len(frs)} sites converted, header at line {sd[0] + 2}")
print(f"TOTAL: {total['has_match']} has_match + {total['count_lines']} count_lines = {sum(total.values())} sites in {len(by_file)} files")
assert sum(total.values()) == art["total"] == 261, "site count mismatch vs artifact"
if __name__ == "__main__":
main()