mirror of
https://github.com/Druthulu/BFM-decomp
synced 2026-09-29 23:24:32 -04:00
feat(tools): work_evidence — assert a tool ACTUALLY DID the work it reports
make tools-health audits DATA integrity (corpus/cdecl/binaries/digest/text) and
nothing audited TOOL BEHAVIOUR -- the gap all four S70 defects fell through. Each
reported success while doing nothing or doing harm, and none would have been found
by reading the source: a wrong instrument returns a plausible NUMBER, not an error.
tools/work_evidence.py, three assertions on OBSERVABLE CONSEQUENCE:
assert_inputs zero readable inputs is a DEFECT, not a zero-yield result. "0 of 0"
is a fact about the harness; "0 of 57" is a fact about the subject.
assert_floor work claiming a compile/gate cannot beat physics -- the ONLY tell on
the pgate defect was a 1-2s wall clock (§402).
assert_effect N claimed successes must show a persistent effect; verification is
not banking (§404).
Self-test is a negative control both directions (11/11): each assertion PASSES the
already-succeeded case and FAILS the known-bad one, and non-strict warns instead of
raising. Wired into `make tools-health` so it cannot rot (R54).
Wiring on the wave critical path:
* parallel_gate: per-worker wall-clock floor; a sub-floor worker is flagged
"BLIND SUSPECT" in the summary line instead of passing as a clean zero.
* gate_stage: the silent `if not draft_fns: return {...}` -- the exact point the
pgate defect flowed through -- is now loud and marks the result `refused`.
* harvest_verify: says at the point of confusion that "verified" is not "banked"
and names gate_stage as the entrypoint that persists.
Negative control: empty drafts dir -> loud + refused. Positive control: a real
2-draft dir still gates normally (drafts:2, no false refusal).
This commit is contained in:
@@ -257,6 +257,9 @@ tools-health:
|
||||
# The cookbook index is DERIVED (R33) and self-asserts its coverage (R32). Stale = agents can't
|
||||
# find documented idioms and re-derive them at full token cost (measured, P30 wave 1).
|
||||
$(VENV_PY) tools/cookbook_index.py --check
|
||||
# Behavioural guards (P31 S70): tools-health audits DATA integrity; these assert that a tool
|
||||
# ACTUALLY DID the work it reports. A guard that is not running is not a guard (R54).
|
||||
$(VENV_PY) tools/work_evidence.py --selftest
|
||||
echo "tools-health: OK — sigs fresh; corpus(+resident) + cdecl + binaries + report(lint+dedup) + cookbook-index all green."
|
||||
|
||||
report:
|
||||
|
||||
+12
-1
@@ -23,6 +23,7 @@ Usage:
|
||||
Prints a JSON summary; importable as run_gate(...)->dict.
|
||||
"""
|
||||
import argparse, fcntl, glob, json, os, re, shutil, subprocess, sys, time
|
||||
import work_evidence
|
||||
import hashlib
|
||||
import glob as _glob
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
@@ -238,7 +239,17 @@ def _run_gate_locked(drafts, binary, src, asm, out, good_sha, propagate, source_
|
||||
draft_fns = sorted(os.path.basename(p)[:-2] for p in
|
||||
glob.glob(os.path.join(REPO, drafts, "*.c")))
|
||||
if not draft_fns:
|
||||
return {"drafts": 0, "banked": 0, "propagated": 0, "near": 0, "failed": 0, "verified": []}
|
||||
# LOUD REFUSAL, NOT A ZERO-YIELD RESULT (P31 S70, §402). This return used to be silent, and
|
||||
# it is the exact point the pgate defect flowed through: a worktree-relative --drafts path
|
||||
# resolved to nothing, so 35 binaries reported "banked 0" with rc=0 in 1-2s each and the run
|
||||
# read as 57 failed drafts. "0 of 0" is a fact about the HARNESS; "0 of 57" is a fact about
|
||||
# the SUBJECT, and reporting the first as the second costs a whole batch (R32/R43).
|
||||
work_evidence.assert_inputs(
|
||||
"gate_stage/%s" % binary, 0, os.path.join(REPO, drafts),
|
||||
hint="a 0-draft gate would otherwise report 'banked 0' as if the drafts had failed.",
|
||||
strict=False)
|
||||
return {"drafts": 0, "banked": 0, "propagated": 0, "near": 0, "failed": 0, "verified": [],
|
||||
"refused": "no .c drafts readable at %s" % os.path.join(REPO, drafts)}
|
||||
# Negative control (P9 / Phase-9): the drafts MUST be INCLUDE_ASM stubs SOMEWHERE in this
|
||||
# binary's sources (main + any _a/_o0 split — a split-only batch is legitimately gated by the
|
||||
# per-split run_gate call, so check the UNION, mirroring lora_grind.open_stubs' glob). If a
|
||||
|
||||
@@ -712,6 +712,16 @@ if _klass:
|
||||
print(' failed by class:', ' '.join('%s=%d' % (k, n) for k, n in sorted(_klass.items())))
|
||||
print('VERIFIED:', ' '.join(verified) or '(none)')
|
||||
print('FAILED :', ' '.join(fn for fn, _ in failed) or '(none)')
|
||||
if verified:
|
||||
# §404 (P31 S70): "verified" means THE BYTE-GATE ACCEPTED IT, not that it is banked. Run standalone
|
||||
# this tool leaves the INCLUDE_ASM stub in place; `gate_stage.py` is the entrypoint that persists,
|
||||
# propagates and commits. S70 read `verified 1 / VERIFIED: func_8002B0B4` here, committed straight
|
||||
# afterwards, and banked NOTHING — the stub is still at src/800.c:18341 and a bank was reported to
|
||||
# the owner that never existed. Say it at the point of confusion, not in a doc nobody re-reads.
|
||||
print(' NOTE: "verified" = the byte-gate ACCEPTED these drafts. If you invoked harvest_verify\n'
|
||||
' directly, they are NOT banked — re-run through `tools/gate_stage.py --drafts <dir>\n'
|
||||
' --binary %s [--commit]` to persist. Confirm with corpus.stubs(), not this line.'
|
||||
% a.binary)
|
||||
# R32 — assert the gate LEFT THE TREE where it found it (plus whatever it banked). `_write(baseline)`
|
||||
# above restores the files render() manages; a carve/isolation can touch files it does not. At 0
|
||||
# verified the tracked diff must be EMPTY, and any residue is a failed draft still spliced in — which
|
||||
|
||||
+14
-2
@@ -51,6 +51,7 @@ to prevent). Adopting is per-BINARY and per-FILE, never a blanket add.
|
||||
plan.json: [{"binary": "ov_SC03_099", "drafts": "/abs/path/to/dir"}, ...]
|
||||
"""
|
||||
import argparse, functools, json, os, re, shutil, subprocess, sys, time
|
||||
import work_evidence
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
@@ -330,6 +331,14 @@ def gate_one(idx, pin, job):
|
||||
return {"binary": binary, "banked": [], "error": "corpus refused in worktree"}
|
||||
r = sh([PY, "tools/gate_stage.py", "--drafts", drafts, "--binary", binary,
|
||||
"--no-propagate", "--source-tag", "pgate"], cwd=wt, timeout=3600)
|
||||
# WALL-CLOCK FLOOR (P31 S70, §402). A worker that reached gate_stage is committed to a full
|
||||
# per-binary build+check; that cannot finish in a couple of seconds. When the drafts path was
|
||||
# unreachable inside the worktree this returned "banked 0" in 1-2s with rc=0 and EVERY other
|
||||
# signal said success — the runtime was the only tell. Non-strict so one odd worker cannot
|
||||
# abort the batch, but the flag rides in the result and the summary prints it (R55).
|
||||
blind = not work_evidence.assert_floor(
|
||||
"pgate/%s" % binary, time.time() - t0, claimed="a full binary build + sha1 check",
|
||||
strict=False)
|
||||
after = stubs_of(wt, binary)
|
||||
banked = sorted(before - after) if after is not None else []
|
||||
files, ovl = {}, None
|
||||
@@ -393,6 +402,7 @@ def gate_one(idx, pin, job):
|
||||
classes.append(part.split(":", 1)[0].strip())
|
||||
cls = " ".join(sorted(set(classes)))
|
||||
return {"binary": binary, "banked": banked, "files": files, "ovl": ovl,
|
||||
"blind_suspect": blind,
|
||||
"secs": round(time.time() - t0, 1),
|
||||
"missing_generated": missing, "classes": cls, "verdicts": verdicts,
|
||||
"rc": r.returncode, "tail": (r.stdout or r.stderr)[-200:] if not banked else ""}
|
||||
@@ -443,8 +453,10 @@ def main():
|
||||
futs[ex.submit(run, job, i % nw)] = job["binary"]
|
||||
for f in as_completed(futs):
|
||||
r = f.result(); results.append(r)
|
||||
print("[pgate] %-14s banked %-3d %s" % (r["binary"], len(r["banked"]),
|
||||
r.get("error") or ("%.0fs" % r.get("secs", 0))), flush=True)
|
||||
print("[pgate] %-14s banked %-3d %s%s" % (r["binary"], len(r["banked"]),
|
||||
r.get("error") or ("%.0fs" % r.get("secs", 0)),
|
||||
" <-- BLIND SUSPECT: finished below the build floor (§402)"
|
||||
if r.get("blind_suspect") else ""), flush=True)
|
||||
finally:
|
||||
if not a.keep:
|
||||
for wt in wts:
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
#!/usr/bin/env python3
|
||||
"""work_evidence.py — assertions that a tool ACTUALLY DID the work it reports (P31 S70).
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
`make tools-health` audits DATA integrity (corpus, cdecl, binaries, digest, text). Nothing audited
|
||||
TOOL BEHAVIOUR — and in one session four tools reported success while doing nothing or doing harm:
|
||||
|
||||
* `parallel_gate` ran `gate_stage` with `cwd=<worktree>` and a RELATIVE `--drafts` path, so the
|
||||
directory did not exist there: 0 drafts found, banked 0, **rc=0**, 1-2s per binary. The same
|
||||
drafts gated in-tree banked 15/16. (§402)
|
||||
* `harvest_verify` reported `verified 1 / VERIFIED: func_8002B0B4`, SHA byte-identical, and left
|
||||
the `INCLUDE_ASM` stub in place — it verifies, it does not bank. A bank was REPORTED that never
|
||||
existed. (§404)
|
||||
* both undo-journals printed full success and left the file corrupted. (§403)
|
||||
* `gater_lane` ledgered drafts as gated that the gate never examined.
|
||||
|
||||
THE COMMON SIGNATURE, and the reason a code review would not have found them: **a wrong instrument
|
||||
returns a plausible NUMBER, not an error.** Every one of these was caught by behaviour — an impossible
|
||||
runtime, a diff after a claimed-success undo, an arithmetic reconciliation that refused to close.
|
||||
|
||||
So the assertions here are about OBSERVABLE CONSEQUENCE, not about internal state:
|
||||
|
||||
assert_inputs a job with zero readable inputs is a DEFECT, not a zero-yield result (R32/R43).
|
||||
"0 of 0 failed" and "0 of 57 failed" are indistinguishable downstream, and the
|
||||
second is a fact about the drafts while the first is a fact about the harness.
|
||||
assert_floor work that claims to have compiled/gated cannot finish faster than a compile. The
|
||||
ONLY tell on the pgate defect was a 1-2s wall clock; every other signal said success.
|
||||
assert_effect a tool claiming N successes must show a PERSISTENT effect. Verification is not
|
||||
banking; a green report with an unchanged tree is the §404 class.
|
||||
|
||||
USAGE — call these where the tool already knows the facts, e.g.
|
||||
|
||||
import work_evidence as we
|
||||
we.assert_inputs("gate_stage", n_drafts, drafts_dir,
|
||||
hint="a 0-draft gate would report 'banked 0' as if the drafts had failed")
|
||||
we.assert_floor("pgate/%s" % binary, elapsed, floor=10.0, claimed="a full binary build+check")
|
||||
we.assert_effect("bank", claimed=len(verified), before=stubs_before, after=stubs_after)
|
||||
|
||||
Each raises `WorkEvidenceError`. Callers that must not abort can pass `strict=False` to get a loud
|
||||
stderr warning and a False return instead — but note R54: a guard that is not running is not a guard,
|
||||
and a warning nobody reads is the same as silence. Prefer the raise.
|
||||
|
||||
Self-test (R39 negative control — every assertion must PASS the good case AND FAIL the bad one):
|
||||
python3 tools/work_evidence.py --selftest
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
FLOOR_BUILD = 10.0 # seconds. Nothing that compiles + links + sha1s a PS1 binary finishes faster.
|
||||
|
||||
|
||||
class WorkEvidenceError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def _fail(msg, strict):
|
||||
if strict:
|
||||
raise WorkEvidenceError(msg)
|
||||
print("[work-evidence] WARNING: %s" % msg, file=sys.stderr)
|
||||
return False
|
||||
|
||||
|
||||
def assert_inputs(what, n_inputs, where, hint="", strict=True):
|
||||
"""A job that found ZERO inputs did not do the work — it never started it.
|
||||
|
||||
The distinction that matters downstream: "processed 0 of 0" is a statement about the HARNESS,
|
||||
"processed 0 of 57" is a statement about the SUBJECT. Reporting the first as the second is how a
|
||||
35-binary batch read as 57 failed drafts (§402)."""
|
||||
if n_inputs and n_inputs > 0:
|
||||
return True
|
||||
return _fail("%s: ZERO readable inputs at %r — refusing to report a zero-yield RESULT for what is "
|
||||
"an empty INPUT.%s" % (what, where, (" " + hint) if hint else ""), strict)
|
||||
|
||||
|
||||
def assert_floor(what, elapsed, floor=FLOOR_BUILD, claimed="", strict=True):
|
||||
"""Work that claims a compile/gate cannot beat physics.
|
||||
|
||||
This is the assertion that would have caught §402 on its own: 35 binaries each 'gated' in 1-2s
|
||||
while a real gate takes 60-120s. Everything else about that run looked like success."""
|
||||
if elapsed is None:
|
||||
return True
|
||||
if elapsed >= floor:
|
||||
return True
|
||||
return _fail("%s: finished in %.1fs, below the %.1fs floor for %s — that is faster than the work "
|
||||
"it claims to have done, so it did not do it." % (what, elapsed, floor,
|
||||
claimed or "this operation"), strict)
|
||||
|
||||
|
||||
def assert_effect(what, claimed, before=None, after=None, effect=None, hint="", strict=True):
|
||||
"""A claim of N successes must have a PERSISTENT consequence.
|
||||
|
||||
Pass either a before/after measurement (e.g. the stub set) or an explicit `effect` boolean. When
|
||||
`claimed` is 0 there is nothing to prove and this is a no-op. §404: `verified 1` with the stub
|
||||
still present is exactly the shape this refuses."""
|
||||
if not claimed:
|
||||
return True
|
||||
if effect is None and before is not None and after is not None:
|
||||
try:
|
||||
effect = len(before) != len(after)
|
||||
except TypeError:
|
||||
effect = before != after
|
||||
if effect:
|
||||
return True
|
||||
return _fail("%s: claims %d success(es) but the tree shows NO persistent effect — a verification "
|
||||
"is not a bank.%s" % (what, claimed, (" " + hint) if hint else ""), strict)
|
||||
|
||||
|
||||
def selftest():
|
||||
"""R39: each assertion must PASS the already-succeeded case and FAIL the known-bad one."""
|
||||
ok = True
|
||||
|
||||
def expect(name, fn, should_raise):
|
||||
nonlocal ok
|
||||
try:
|
||||
fn()
|
||||
raised = False
|
||||
except WorkEvidenceError:
|
||||
raised = True
|
||||
good = raised == should_raise
|
||||
ok &= good
|
||||
print(" [%s] %-42s %s" % ("PASS" if good else "FAIL", name,
|
||||
"raised" if raised else "no raise"))
|
||||
|
||||
expect("inputs: 57 drafts -> accept",
|
||||
lambda: assert_inputs("t", 57, "/d"), False)
|
||||
expect("inputs: 0 drafts -> REFUSE",
|
||||
lambda: assert_inputs("t", 0, "/d"), True)
|
||||
expect("inputs: None -> REFUSE",
|
||||
lambda: assert_inputs("t", None, "/d"), True)
|
||||
|
||||
expect("floor: 102s build -> accept",
|
||||
lambda: assert_floor("t", 102.0), False)
|
||||
expect("floor: 1.4s 'build' -> REFUSE (the §402 tell)",
|
||||
lambda: assert_floor("t", 1.4), True)
|
||||
expect("floor: elapsed unknown -> accept",
|
||||
lambda: assert_floor("t", None), False)
|
||||
|
||||
expect("effect: claimed 0 -> accept (nothing to prove)",
|
||||
lambda: assert_effect("t", 0, before={1, 2}, after={1, 2}), False)
|
||||
expect("effect: claimed 1, stub set SHRANK -> accept",
|
||||
lambda: assert_effect("t", 1, before={1, 2}, after={2}), False)
|
||||
expect("effect: claimed 1, tree UNCHANGED -> REFUSE (the §404 tell)",
|
||||
lambda: assert_effect("t", 1, before={1, 2}, after={1, 2}), True)
|
||||
expect("effect: explicit effect=True -> accept",
|
||||
lambda: assert_effect("t", 3, effect=True), False)
|
||||
|
||||
# non-strict must WARN and return False, never raise
|
||||
quiet = assert_inputs("t", 0, "/d", strict=False)
|
||||
ok &= (quiet is False)
|
||||
print(" [%s] non-strict returns False instead of raising" % ("PASS" if quiet is False else "FAIL"))
|
||||
|
||||
print("work_evidence selftest: %s" % ("OK" if ok else "FAILED"))
|
||||
return 0 if ok else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(selftest() if "--selftest" in sys.argv else selftest())
|
||||
Reference in New Issue
Block a user