fix(wake): #927 enqueue TOCTOU — move stale-tmp cleanup off the hot enqueue path (no concurrent in-flight-write clobber) (#928)
Co-authored-by: jason.woltje <jason@diversecanvas.com> Co-committed-by: jason.woltje <jason@diversecanvas.com>
This commit was merged in pull request #928.
This commit is contained in:
@@ -170,11 +170,27 @@ echo "== T5: atomic write-tmp+rename survives crash mid-write =="
|
||||
printf '%s\n' "$drained" | jq -e . >/dev/null 2>&1 || fail_msg "T5: drain returned non-JSON — garbage temp corrupted the store"
|
||||
printf '%s\n' "$drained" | stream_any '.observed_seq==1' || fail_msg "T5: committed entry (seq 1) lost after simulated crash"
|
||||
printf '%s\n' "$drained" | grep -q 999 && fail_msg "T5: uncommitted garbage (seq 999) leaked into the live store"
|
||||
# A subsequent real mutation must succeed and clean the stale temp.
|
||||
# A subsequent real mutation must succeed DESPITE the stale temp present.
|
||||
"$STORE" enqueue --seq 2 --class actionable --locators '{"issue":2}' >/dev/null || fail_msg "T5: enqueue after crash temp failed"
|
||||
[ -e "$STATE_DIR/.wake.tmp.crash12" ] && fail_msg "T5: stale crash temp not cleaned by next mutation"
|
||||
# #927: the enqueue HOT PATH must NOT reap tmp files — an unconditional delete
|
||||
# there clobbered a concurrent enqueue's LIVE in-flight tmp mid-write (spurious
|
||||
# "durable pending write FAILED"). So a FRESH stale tmp is intentionally left
|
||||
# untouched by an enqueue; reaping it on the hot path is exactly the bug.
|
||||
[ -e "$STATE_DIR/.wake.tmp.crash12" ] || fail_msg "T5: the enqueue hot path must NOT delete tmp files off its own write (that hot-path reap was the #927 clobber)"
|
||||
d2="$("$STORE" drain)"
|
||||
[ "$(printf '%s\n' "$d2" | grep -c .)" = "2" ] || fail_msg "T5: store not healthy after crash+recovery (expected 2 entries)"
|
||||
# Bounded accumulation is preserved via a MAINTENANCE reap (store.sh init /
|
||||
# detector tick), age-scoped so it only removes DEMONSTRABLY-orphaned tmps
|
||||
# (older than any plausible in-flight write) and never a live one. Age the
|
||||
# crash temp into the past so it is unambiguously orphaned, then run the
|
||||
# maintenance path and confirm it is reaped.
|
||||
touch -d '1 hour ago' "$STATE_DIR/.wake.tmp.crash12" 2>/dev/null ||
|
||||
touch -t "$(date -d '1 hour ago' +%Y%m%d%H%M.%S 2>/dev/null || echo 197001010000)" "$STATE_DIR/.wake.tmp.crash12" 2>/dev/null || true
|
||||
"$STORE" init >/dev/null 2>&1 || fail_msg "T5: store.sh init (maintenance) failed"
|
||||
[ -e "$STATE_DIR/.wake.tmp.crash12" ] && fail_msg "T5: maintenance reap (store.sh init) must remove a demonstrably-orphaned stale temp (bounded accumulation)"
|
||||
d3="$("$STORE" drain)"
|
||||
[ "$(printf '%s\n' "$d3" | grep -c .)" = "2" ] || fail_msg "T5: store not healthy after maintenance reap (expected 2 entries)"
|
||||
printf '%s\n' "$d3" | jq -e . >/dev/null 2>&1 || fail_msg "T5: drain returned non-JSON after maintenance reap"
|
||||
) && ok
|
||||
|
||||
echo "== T5b: _atomic_write is tmp+rename (killing the writer mid-write leaves the target intact) =="
|
||||
|
||||
Reference in New Issue
Block a user