Files
stack/packages/mosaic/framework/tools/wake/validate-973/convert-973.py
T
Mos 4fb44f6345
ci/woodpecker/push/publish Pipeline was successful
ci/woodpecker/push/ci Pipeline was successful
fix(wake): three-valued grep verdicts — has_match/count_lines across all ten suites (closes #973) (#983)
2026-07-31 08:39:04 +00:00

132 lines
5.1 KiB
Python

#!/usr/bin/env python3
"""convert-973.py — mechanical #973 site conversion, driven by the frozen
denominator artifact (denominator-089615f.json), never by ad-hoc grepping.
For every row in the artifact it FIRST verifies the worktree line still says
what the artifact froze (exact match after whitespace strip, artifact-side
truncation as prefix match, or the normalized suite-summary template), and
aborts before touching anything on the first verification failure — a
conversion applied to a line the denominator did not measure would be the
umbrella defect wearing the converter's clothes.
Transforms (behaviour-preserving; verdict semantics unchanged on grep rc 0/1):
E-count-capture `grep -c ARGS` -> `count_lines ARGS` (helper adds -c)
all other forms `grep ARGS` -> `has_match ARGS` (drop-in)
env prefixes (`LC_ALL=C grep`) are kept — the prefix reaches the grep child
through the function (microtest C8).
The two multi-grep pipeline sites (digest-quarantine:560, install:420) convert
BOTH greps: each is measurement-bearing, and under pipefail an rc=2 in the
left element is masked by an rc=1 in the right — the same defect one pipe
deeper. They stay ONE denominator site each (one coordinate); the ledger
records helper calls, so those coordinates appear twice per execution and the
equality check compares SETS of coordinates.
Finally each suite gains three header lines directly after its SCRIPT_DIR
assignment (source + wake_assert_init + comment), shifting every site by +3
lines exactly; the validation harness maps base coordinates accordingly.
"""
import json
import re
import sys
from pathlib import Path
HERE = Path(__file__).resolve().parent
WAKE = HERE.parent
ART = HERE / "denominator-089615f.json"
RX_GREP_TOKEN = re.compile(r"(^|[^A-Za-z0-9_.-])grep([^A-Za-z0-9_.-]|$)")
HEADER = [
"# #973: three-valued grep assertion helpers (has_match/count_lines); init saves real stderr for abort loudness.\n",
"# shellcheck disable=SC1091\n",
'. "$SCRIPT_DIR/_wake-common.sh" && wake_assert_init\n',
]
MULTI_GREP_CONVERT_BOTH = {
("test-wake-digest-quarantine.sh", 560),
("test-wake-install.sh", 420),
}
SUMMARY_TEMPLATE_MARK = '$(grep -c . "$FAILFILE")'
def verify(row, actual):
a = actual.strip()
t = row["text"].strip()
if a == t:
return True
if t and a.startswith(t): # artifact-side truncation
return True
# normalized suite-summary template rows
if t.startswith('echo "wake ') and SUMMARY_TEMPLATE_MARK in actual and a.startswith('echo "wake '):
return True
return False
def convert_line(row, line):
key = (row["file"], row["line"])
n_grep = len(RX_GREP_TOKEN.findall(line))
if key in MULTI_GREP_CONVERT_BOTH:
assert n_grep == 2, f"{key}: expected 2 grep tokens, found {n_grep}"
else:
assert n_grep == 1, f"{key}: expected 1 grep token, found {n_grep}: {line!r}"
if row["form"].startswith("E"):
assert line.count("grep -c ") == 1, f"{key}: E row without single 'grep -c ': {line!r}"
return line.replace("grep -c ", "count_lines ", 1)
def repl(m):
return m.group(1) + "has_match" + m.group(2)
count = 2 if key in MULTI_GREP_CONVERT_BOTH else 1
return RX_GREP_TOKEN.sub(repl, line, count=count)
def main():
art = json.loads(ART.read_text())
rows = art["rows"]
by_file = {}
for r in rows:
by_file.setdefault(r["file"], []).append(r)
# pass 1: verify every row before touching any file
bad = 0
texts = {}
for f, frs in by_file.items():
lines = (WAKE / f).read_text().split("\n")
texts[f] = lines
for r in frs:
if not verify(r, lines[r["line"] - 1]):
bad += 1
print(f"VERIFY-FAIL {f}:{r['line']}\n artifact: {r['text']!r}\n worktree: {lines[r['line'] - 1]!r}")
if bad:
sys.exit(f"ABORT: {bad} row(s) failed verification; nothing was modified.")
# pass 2: convert + insert header
total = {"has_match": 0, "count_lines": 0}
for f, frs in sorted(by_file.items()):
lines = texts[f]
for r in frs:
i = r["line"] - 1
new = convert_line(r, lines[i])
assert new != lines[i], f"{f}:{r['line']}: no-op conversion"
lines[i] = new
total["count_lines" if r["form"].startswith("E") else "has_match"] += 1
# header insertion after the SCRIPT_DIR= line
sd = [i for i, ln in enumerate(lines) if ln.startswith('SCRIPT_DIR="$(')]
assert len(sd) == 1, f"{f}: expected exactly one SCRIPT_DIR line, found {len(sd)}"
first_site = min(r["line"] for r in frs) - 1
assert sd[0] < first_site, f"{f}: SCRIPT_DIR line {sd[0] + 1} not before first site {first_site + 1}"
lines[sd[0] + 1 : sd[0] + 1] = [h.rstrip("\n") for h in HEADER]
(WAKE / f).write_text("\n".join(lines))
print(f"{f}: {len(frs)} sites converted, header at line {sd[0] + 2}")
print(f"TOTAL: {total['has_match']} has_match + {total['count_lines']} count_lines = {sum(total.values())} sites in {len(by_file)} files")
assert sum(total.values()) == art["total"] == 261, "site count mismatch vs artifact"
if __name__ == "__main__":
main()