feat(phase-23): local GPU serving + prompt-fix + corpus-v3 (macro+struct mining)

LM Studio was ejected, so serving is ours now: tools/serve_local.py serves base+LoRA via
Unsloth (.venv-train cu128) as an OpenAI endpoint — no llama.cpp build (this CPU has no
AVX-512, which SIGILLs the prebuilt llama-cpp-python CUDA wheels). api_draft/lora_grind hit
it unchanged.

PROMPT FIX (api_draft.LEAN_SYS + format_finetune.SYS, kept in sync): 'translate EVERY
instruction, never an empty body' — a prompt test took the small-leaf band 0/3 -> 2/3 MATCH
(the v2 corpus overfit an empty void f(void){} leaf pattern). Validated end-to-end: a fresh
ov_SC01_001 batch banked 3 via the local server + better prompt.

CORPUS-V3 (export_pairs + format_finetune):
- export_pairs now ALSO mines the 1623 DEFINE_func macro bodies in engine_core.h (the shared
  setters/return-const/dispatchers extract_defs never saw -> 96.6% of v2 was overlay-unique,
  the root of the empty-leaf overfit). Corpus 1312 inline -> 2891 (1312 inline + 1579 macros).
- format_finetune inlines engine_types.h structs in the compile-filter so struct-using bodies
  are KEPT not dropped: train 2534/2591 (97.8%) compile standalone (v2 was 1111 total).
This commit is contained in:
Drew T
2026-06-30 01:46:54 -06:00
parent 5c44a317b3
commit 066a2e5c8f
4 changed files with 154 additions and 5 deletions
+10 -2
View File
@@ -111,11 +111,19 @@ Now write byte-matching C for {t['name']}:
- Reply with ONLY one ```c block."""
# LEAN mode — the SAME prompt shape as tools/format_finetune.py (train/inference must match).
# LEAN mode — keep in sync with tools/format_finetune.py (train/inference must match). The
# "translate EVERY instruction / never-empty" clause was added 2026-06-30 after a prompt test took
# the small-leaf band 0/3 -> 2/3 MATCH (the v2 corpus overfit an empty `void f(void){}` leaf pattern;
# the instruction it most often dropped was the return value / a store). MIRROR this in format_finetune
# before retraining corpus-v3, else train/inference drift.
LEAN_SYS = ("You are an expert at MATCHING decompilation for MIPS (PSX, gcc-2.7.2 -O2 -G0 -mips1 -mcpu=3000 "
"-msoft-float + maspsx). Given a function's target assembly, output C that the pinned toolchain "
"compiles to BYTE-IDENTICAL machine code. The types u8/u16/u32/s8/s16/s32/f32/s64/u64/f64 are "
"predefined (common.h). Output ONLY the C (the function definition + any externs it needs).")
"predefined (common.h). Output ONLY the C (the function definition + any externs it needs). "
"Translate EVERY instruction — NEVER output an empty body. A `jr $ra` with `addiu $v0,$zero,N` "
"in its delay slot is `return N;`; a `sw/sh/sb $aK,off($a0)` is a store "
"`*(T*)((u8*)arg0+off)=argK;` (T=s32/s16/s8); a `lw/lh/lb` is a load. Produce C whose compiled "
"output IS the shown instructions.")
# Bridge: real OPEN stubs are splat .s (headers, 3-field comment, spaced operands, resolved jal); the
+37 -1
View File
@@ -100,6 +100,34 @@ def extract_defs(src_text):
return out
def extract_macro_defs(text):
"""corpus-v3: mine the shared `#define DEFINE_func_XXXX() <full def>` macros from engine_core.h.
These hold the byte-matched SHARED engine functions (setters, return-const, dispatchers) that
extract_defs() can't see — it matches only column-0 inline `func_X(){...}` defs, never the macro
instantiations. The model trained without them (96.6% of corpus-v2 was overlay-unique inline defs),
so it OVERFIT an empty `void f(void){}` leaf pattern and drafts trivial setters/returns empty.
Mining the ~1600 macro bodies feeds exactly the missing variety. Inline `/* asm */` annotations are
stripped (the model's output is clean C; train/inference style must match)."""
out, lines, i = {}, text.split('\n'), 0
while i < len(lines):
m = re.match(r'\s*#define DEFINE_(func_[0-9A-Fa-f]+)\(\)\s*\\\s*$', lines[i])
if not m:
i += 1
continue
fn, i, body = m.group(1), i + 1, []
while i < len(lines):
ln = lines[i].rstrip()
i += 1
cont = ln.endswith('\\')
body.append(ln[:-1].rstrip() if cont else ln)
if not cont:
break
c = re.sub(r'/\*.*?\*/', '', '\n'.join(body)) # drop inline asm-annotation comments
out[fn] = '\n'.join(l.rstrip() for l in c.split('\n') if l.strip())
return out
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument('--out', default='datasets/match_pairs')
@@ -122,7 +150,15 @@ def main():
region = os.path.relpath(os.path.join(root, f), os.path.join(REPO, 'src'))[:-2]
for fn, body in extract_defs(open(os.path.join(root, f)).read()).items():
defs.setdefault(fn, (body, region))
print(' %d func defs in src' % len(defs), file=sys.stderr)
# corpus-v3: ALSO mine the shared DEFINE_func macro bodies (engine_core.h) — the setters/return-const/
# dispatchers the model is blind to (extract_defs sees only inline col-0 defs, not macro instantiations).
n_inline = len(defs)
ec = os.path.join(REPO, 'src/shared/engine_core.h')
if os.path.exists(ec):
for fn, body in extract_macro_defs(open(ec).read()).items():
defs.setdefault(fn, (body, 'shared'))
print(' %d func defs in src (%d inline + %d shared macros)' % (
len(defs), n_inline, len(defs) - n_inline), file=sys.stderr)
pairs, skipped = [], {'no_asm': 0}
for fn, (body, region) in sorted(defs.items()):
+22 -2
View File
@@ -27,8 +27,22 @@ _ASF = '-Iinclude -march=r3000 -mtune=r3000 -no-pad-sections -O1 -G0'.split()
_TD = re.compile(r'^[ \t]*typedef\b.*\b(u8|u16|u32|u64|s8|s16|s32|s64|f32|f64)[ \t]*;[ \t]*\n', re.M)
_ETYPES = None
def _engine_types():
"""src/shared/engine_types.h content (the shared Actor-class struct/union/typedefs), cached."""
global _ETYPES
if _ETYPES is None:
p = os.path.join(REPO, 'src/shared/engine_types.h')
_ETYPES = (open(p).read() + '\n') if os.path.exists(p) else ''
return _ETYPES
def compiles(c):
src = '#include "common.h"\n' + _TD.sub('', c)
# corpus-v3: inline the shared struct/typedefs so struct-using macro bodies COMPILE and are KEPT
# (else the filter drops every fn that touches an Actor-class field). common.h provides the scalars.
src = '#include "common.h"\n' + _engine_types() + _TD.sub('', c)
wd = os.path.join(REPO, '.run/_ft_cc'); os.makedirs(wd, exist_ok=True)
open(os.path.join(wd, 't.c'), 'w').write(src)
p = subprocess.run([_CPP] + _CPPF + ['.run/_ft_cc/t.c'], capture_output=True, cwd=REPO)
@@ -40,10 +54,16 @@ def compiles(c):
p = subprocess.run([_AS] + _ASF + ['-o', '.run/_ft_cc/t.o'], input=p.stdout, capture_output=True, cwd=REPO)
return p.returncode == 0
# MUST match api_draft.LEAN_SYS exactly (train/inference align). The "translate EVERY instruction /
# never-empty" clause (added 2026-06-30) fixes the empty-leaf overfit (prompt test: small-leaf 0/3 -> 2/3).
SYS = ("You are an expert at MATCHING decompilation for MIPS (PSX, gcc-2.7.2 -O2 -G0 -mips1 -mcpu=3000 "
"-msoft-float + maspsx). Given a function's target assembly, output C that the pinned toolchain "
"compiles to BYTE-IDENTICAL machine code. The types u8/u16/u32/s8/s16/s32/f32/s64/u64/f64 are "
"predefined (common.h). Output ONLY the C (the function definition + any externs it needs).")
"predefined (common.h). Output ONLY the C (the function definition + any externs it needs). "
"Translate EVERY instruction — NEVER output an empty body. A `jr $ra` with `addiu $v0,$zero,N` "
"in its delay slot is `return N;`; a `sw/sh/sb $aK,off($a0)` is a store "
"`*(T*)((u8*)arg0+off)=argK;` (T=s32/s16/s8); a `lw/lh/lb` is a load. Produce C whose compiled "
"output IS the shown instructions.")
def user_msg(asm):
+85
View File
@@ -0,0 +1,85 @@
#!/usr/bin/env python3
"""serve_local.py — minimal OpenAI-compatible server for the fine-tuned BFM matcher, on the GPU.
Loads the base Qwen2.5-Coder-7B (4-bit) + a LoRA adapter via Unsloth — the SAME .venv-train stack
that trained it, so no llama.cpp build and no SIGILL-prone prebuilt wheel (this host's CPU has no
AVX-512, which crashes the generic llama-cpp-python CUDA wheels). Serves /v1/chat/completions +
/v1/models so tools/api_draft.py + tools/lora_grind.py hit it UNCHANGED.
.venv-train/bin/python tools/serve_local.py --adapter models/bfm-match-7b --name bfm-match-7b-v2 --port 1234
API_BASE=http://127.0.0.1:1234/v1 MODEL=bfm-match-7b-v2 ... tools/lora_grind.py ...
On-demand only (no daemon); SIGTERM/Ctrl-C to stop. One GPU generation at a time (a lock serializes
— lora_grind drafts sequentially anyway). The requested `model` field is ignored; the loaded adapter
is served (so MODEL can be any non-empty string).
"""
import argparse, json, threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--adapter", default="models/bfm-match-7b")
ap.add_argument("--name", default="bfm-match-local")
ap.add_argument("--port", type=int, default=1234)
ap.add_argument("--host", default="127.0.0.1")
ap.add_argument("--max-seq", type=int, default=4096)
a = ap.parse_args()
from unsloth import FastLanguageModel
import torch
print(f"[serve] loading {a.adapter} (base+adapter, 4-bit, GPU)...", flush=True)
model, tok = FastLanguageModel.from_pretrained(a.adapter, max_seq_length=a.max_seq, load_in_4bit=True)
FastLanguageModel.for_inference(model)
lock = threading.Lock()
print(f"[serve] model ready: {a.name}", flush=True)
def generate(messages, max_tokens, temperature):
prompt = tok.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
inputs = tok(prompt, return_tensors="pt").to("cuda")
do_sample = bool(temperature and temperature > 0)
with lock, torch.no_grad():
out = model.generate(**inputs, max_new_tokens=max_tokens, do_sample=do_sample,
temperature=(temperature or 1.0), pad_token_id=tok.eos_token_id)
return tok.decode(out[0][inputs["input_ids"].shape[1]:], skip_special_tokens=True)
class H(BaseHTTPRequestHandler):
def log_message(self, *args):
pass
def _json(self, code, obj):
b = json.dumps(obj).encode()
self.send_response(code)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(b)))
self.end_headers()
self.wfile.write(b)
def do_GET(self):
if self.path.rstrip("/").endswith("/models"):
self._json(200, {"object": "list", "data": [{"id": a.name, "object": "model"}]})
else:
self._json(404, {"error": "not found"})
def do_POST(self):
if not self.path.endswith("/chat/completions"):
self._json(404, {"error": "not found"})
return
n = int(self.headers.get("Content-Length", 0))
req = json.loads(self.rfile.read(n) or b"{}")
try:
txt = generate(req.get("messages", []), int(req.get("max_tokens", 1024)),
float(req.get("temperature", 0.2)))
self._json(200, {"id": "cmpl", "object": "chat.completion", "model": a.name,
"choices": [{"index": 0, "finish_reason": "stop",
"message": {"role": "assistant", "content": txt}}]})
except Exception as e:
self._json(500, {"error": str(e)})
srv = ThreadingHTTPServer((a.host, a.port), H)
print(f"[serve] {a.name} serving on http://{a.host}:{a.port}/v1", flush=True)
srv.serve_forever()
if __name__ == "__main__":
main()