Files
BFM-decomp/tools/public_rewrite/scrub.py
T
Drew T fca3839049 feat(phase-33): C1 the history-rewrite tooling — tools/public_rewrite/ (git-filter-repo 2.47.0), PROVEN by two trial rewrites on a scratch bare clone
- hash_dict (4,420 commit objects -> commit:NNNN / twin / orphan; 150,280 prefixes; 0 ambiguous; 0 collisions with 1,480
  cited content hashes; scratch mailmap), scrub (12/12 self-test; HEAD sample 731 distinct tokens == git's own lookup),
  gate_scan (+ expected_offenders.txt fixture: the R39 negative control, PASS over 16.8 GB in 2 m 25 s; rom_blob_ids =
  content/signature hits U purge-path blobs not shared with any other path), run_filter (module API; --force only for the
  deleted tag), verify_rewrite (pairwise proof + no purge path survives + pruned set == derived purge-only set),
  build_commit_map (0 old hashes asserted; unchanged commits exempted), resolve_tokens, absent_scan (positive control on the
  current repo: FAIL 82,362), probe_github.sh (skips unchanged commits)
- trial #1 found two defects the plan had not foreseen: the EMPTY blob in the strip list (--strip-blobs-with-ids then
  undid every "file emptied" change in history and pruned a restore commit) and the byte-identical Initial commit keeping
  its hash; both fixed with controls
- trial #2: filter 269 s, exactly 1 pruned, verify 4,029 pairs / 0 failures, map 4,030 rows, absent_scan + gate PASS on
  the clone, 1,231 tokens resolved at a trial tip (residue 7: the pruned commit x3, orphans x4), aggressive repack 500 -> 80 MB
- SETUP P33 C1 section + rows (R21); runbook §3/§5/§6/§10 measured; requirements-python.txt; CURRENT_PHASE -> NEXT = C2
2026-09-06 22:37:10 -06:00

212 lines
9.9 KiB
Python

#!/usr/bin/env python3
"""scrub.py — THE one scrub function of the public-flip rewrite (P33 C1), and its sample / self-test.
tools/public_rewrite/scrub.py --test # known-true cases (R39)
tools/public_rewrite/scrub.py --sample [--rev HEAD] # every tracked blob of a tree: hits, files, per-length counts,
# the 7-char replacements in context, an INDEPENDENT oracle
tools/public_rewrite/scrub.py --file IN --out OUT # scrub one file (debugging)
What it does to a text blob or a commit message:
* every hex token 7..40 chars long that is a prefix of exactly one OLD commit hash (the dictionary) becomes the
inert token `commit:NNNN` (main ordinal), `commit:orphan-K`, or `commit:amb-K` for an ambiguous prefix; tokens
that are also prefixes of a cited CONTENT hash are left alone (the exclusion set); anything else is untouched;
* every personal e-mail address (from the scratch mailmap) becomes the noreply address;
* for MESSAGES only: `Claude-Session: …` trailer lines are dropped (60 on main, 130 over all refs at S87).
Binary blobs (a NUL in the first 8 KiB) are never touched. The function is idempotent: scrub(scrub(x)) == scrub(x).
"""
import argparse
import collections
import re
import sys
import time
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parent))
import common as C # noqa: E402
class Scrubber:
def __init__(self, dictionary=None, personal_emails=None):
d = dictionary if dictionary is not None else C.load_dict()
self.commits = d["commits"]
self.index = {}
for h in self.commits:
for p in C.prefixes_of(h):
self.index.setdefault(p, h)
self.ambiguous = {p: i + 1 for i, p in enumerate(d["ambiguous"])}
self.excluded = set(d["excluded"])
self.emails = [e.encode() for e in (personal_emails if personal_emails is not None
else C.personal_emails_from_mailmap())]
self.noreply = C.NOREPLY_EMAIL.encode()
self.stats = collections.Counter()
self.replaced = collections.Counter() # token -> count (for the sample)
def token_for(self, tok):
"""The replacement for one hex token (bytes) or None."""
t = tok.decode()
if t in self.excluded:
self.stats["excluded"] += 1
return None
if t in self.ambiguous:
self.stats["ambiguous"] += 1
return b"commit:amb-%d" % self.ambiguous[t]
h = self.index.get(t)
if h is None:
self.stats["unresolved"] += 1
return None
e = self.commits[h]
if e["kind"] in ("main", "twin"):
return b"commit:%04d" % e["ord"]
return b"commit:orphan-%d" % e["orphan"]
def _sub(self, m):
r = self.token_for(m.group(0))
if r is None:
return m.group(0)
self.stats["replaced"] += 1
self.stats[f"replaced_len{len(m.group(0))}"] += 1
self.replaced[m.group(0)] += 1
return r
def scrub_text(self, data):
if C.is_binary(data):
self.stats["binary"] += 1
return data
out = C.HEX_RE.sub(self._sub, data)
for e in self.emails:
if e in out:
self.stats["emails"] += out.count(e)
out = out.replace(e, self.noreply)
return out
def scrub_message(self, msg):
out, n = C.TRAILER_RE.subn(b"", msg)
self.stats["trailers"] += n
out = self.scrub_text(out)
if n:
out = out.rstrip(b"\n") + b"\n"
return out
# ---------------------------------------------------------------- self-test (known-true cases)
def selftest():
fake_main = "0123456789abcdef0123456789abcdef01234567"
fake_twin = "fedcba9876543210fedcba9876543210fedcba98"
fake_orphan = "1111111111111111111111111111111111111111"
content = "abcdef0123456789abcdef0123456789abcdef01" # a "manifest" sha1: prefix abcdef0 collides with nothing
d = {"commits": {fake_main: {"kind": "main", "ord": 12}, fake_twin: {"kind": "twin", "ord": 3},
fake_orphan: {"kind": "orphan", "orphan": 2}},
"ambiguous": [], "excluded": sorted(C.prefixes_of(content))}
s = Scrubber(d, personal_emails=["someone@example.com"])
cases = [
(b"see 0123456789a for the fix", b"see commit:0012 for the fix", "9-char main prefix"),
(b"(" + fake_main.encode() + b")", b"(commit:0012)", "full 40-char main hash"),
(b"twin fedcba98 here", b"twin commit:0003 here", "8-char twin prefix -> the twin's ordinal"),
(b"lost 1111111 commit", b"lost commit:orphan-2 commit", "orphan"),
(b"func_800D128C and 0x800d128c stay", b"func_800D128C and 0x800d128c stay", "word-embedded hex untouched"),
(b"abcdef0123 is a manifest sha1 prefix", b"abcdef0123 is a manifest sha1 prefix", "excluded content-hash prefix"),
(b"deadbeefcafe is not a commit", b"deadbeefcafe is not a commit", "unknown hex untouched"),
(b"012345 too short", b"012345 too short", "6 chars never match"),
(b"mail someone@example.com now", b"mail " + C.NOREPLY_EMAIL.encode() + b" now", "personal address -> noreply"),
(b"\0binary 0123456789a", b"\0binary 0123456789a", "binary untouched"),
]
bad = 0
for src, want, name in cases:
got = s.scrub_text(src)
ok = got == want
bad += not ok
print(f" {'ok ' if ok else 'BAD'} {name}: {got!r}")
msg = b"subject 0123456789a\n\nbody\n\nClaude-Session: https://example/x\n"
got = s.scrub_message(msg)
want = b"subject commit:0012\n\nbody\n"
ok = got == want
bad += not ok
print(f" {'ok ' if ok else 'BAD'} message: trailer dropped + hash tokenized: {got!r}")
idem = s.scrub_text(s.scrub_text(b"x 0123456789a y fedcba98")) == s.scrub_text(b"x 0123456789a y fedcba98")
bad += not idem
print(f" {'ok ' if idem else 'BAD'} idempotent")
print(f"scrub --test: {'OK' if not bad else f'{bad} FAILED'}")
return 1 if bad else 0
# ---------------------------------------------------------------- the sample over one tree (R37: measure first)
def sample(rev, repo):
s = Scrubber()
entries = [ln.split("\t")[1] for ln in C.git(["ls-tree", "-r", rev], repo).splitlines()]
ids = {ln.split("\t")[1]: ln.split()[2] for ln in C.git(["ls-tree", "-r", rev], repo).splitlines()}
cf = C.CatFile(repo)
t0 = time.time()
files_hit, nbytes, ctx7 = 0, 0, []
all_tokens = collections.Counter()
for path in entries:
_, _, data = cf.get(ids[path])
if data is None or C.is_binary(data):
continue
nbytes += len(data)
before = s.stats["replaced"]
out = s.scrub_text(data)
if s.stats["replaced"] != before:
files_hit += 1
for m in C.HEX_RE.finditer(data):
all_tokens[m.group(0)] += 1
if len(m.group(0)) == 7 and s.token_for(m.group(0)):
a, b = max(0, m.start() - 40), min(len(data), m.end() + 40)
ctx7.append(f"{path}: …{data[a:b].decode('utf-8', 'replace')}…".replace("\n", "⏎"))
cf.close()
wall = time.time() - t0
# INDEPENDENT oracle: which distinct tokens does GIT ITSELF resolve to a commit object?
distinct = sorted(t.decode() for t in all_tokens)
resolved_by_git = set()
for i in range(0, len(distinct), 2000):
chunk = distinct[i:i + 2000]
out = C.git(["cat-file", "--batch-check"], repo, input="\n".join(f"{t}^{{commit}}" for t in chunk) + "\n", check=False)
for tok, ln in zip(chunk, out.splitlines()):
if " commit " in ln:
resolved_by_git.add(tok)
ours = {t.decode() for t in s.replaced}
only_git = sorted(resolved_by_git - ours)
only_ours = sorted(ours - resolved_by_git)
per_len = {k: v for k, v in sorted(s.stats.items()) if k.startswith("replaced_len")}
print(f"scrub --sample {rev}: {len(entries)} tracked paths, {nbytes / 1e6:.0f} MB of text scanned in {wall:.1f} s "
f"({nbytes / 1e6 / max(wall, 1e-9):.0f} MB/s); {s.stats['replaced']} replacements in {files_hit} files; "
f"{len(ours)} distinct tokens replaced; excluded {s.stats['excluded']}, ambiguous {s.stats['ambiguous']}, "
f"unresolved hex tokens {s.stats['unresolved']}, e-mail replacements {s.stats['emails']}")
print(f" per length: {per_len}")
print(f" independent oracle (git cat-file on every distinct hex token, {len(distinct)} tokens): git resolves "
f"{len(resolved_by_git)} to commits; ours {len(ours)}; only-git {len(only_git)} (expected: the excluded "
f"content-hash prefixes, if any); only-ours {len(only_ours)} (MUST be 0)")
for t in only_git[:10]:
print(f" only-git: {t} {'(excluded)' if t in s.excluded else '(!! dictionary gap)'}")
for t in only_ours[:10]:
print(f" only-ours: {t} !!")
print(f" 7-char replacements in context ({len(ctx7)} — READ them, R63):")
for c in ctx7[:60]:
print(f" {c[:200]}")
return 1 if only_ours or any(t not in s.excluded for t in only_git) else 0
def main(argv):
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--test", action="store_true")
ap.add_argument("--sample", action="store_true")
ap.add_argument("--rev", default="HEAD")
ap.add_argument("--repo", default=str(C.REPO))
ap.add_argument("--file")
ap.add_argument("--out")
a = ap.parse_args(argv)
if a.test:
return selftest()
if a.sample:
return sample(a.rev, a.repo)
if a.file:
s = Scrubber()
data = open(a.file, "rb").read()
out = s.scrub_text(data)
open(a.out or (a.file + ".scrubbed"), "wb").write(out)
print(f"scrub: {dict(s.stats)}")
return 0
ap.error("one of --test / --sample / --file")
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))