From 066a2e5c8f61a7f78a1240c1e54fa58caae4e21e Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Tue, 30 Jun 2026 01:46:54 -0600 Subject: [PATCH] feat(phase-23): local GPU serving + prompt-fix + corpus-v3 (macro+struct mining) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit LM Studio was ejected, so serving is ours now: tools/serve_local.py serves base+LoRA via Unsloth (.venv-train cu128) as an OpenAI endpoint — no llama.cpp build (this CPU has no AVX-512, which SIGILLs the prebuilt llama-cpp-python CUDA wheels). api_draft/lora_grind hit it unchanged. PROMPT FIX (api_draft.LEAN_SYS + format_finetune.SYS, kept in sync): 'translate EVERY instruction, never an empty body' — a prompt test took the small-leaf band 0/3 -> 2/3 MATCH (the v2 corpus overfit an empty void f(void){} leaf pattern). Validated end-to-end: a fresh ov_SC01_001 batch banked 3 via the local server + better prompt. CORPUS-V3 (export_pairs + format_finetune): - export_pairs now ALSO mines the 1623 DEFINE_func macro bodies in engine_core.h (the shared setters/return-const/dispatchers extract_defs never saw -> 96.6% of v2 was overlay-unique, the root of the empty-leaf overfit). Corpus 1312 inline -> 2891 (1312 inline + 1579 macros). - format_finetune inlines engine_types.h structs in the compile-filter so struct-using bodies are KEPT not dropped: train 2534/2591 (97.8%) compile standalone (v2 was 1111 total). --- tools/api_draft.py | 12 +++++- tools/export_pairs.py | 38 +++++++++++++++++- tools/format_finetune.py | 24 +++++++++++- tools/serve_local.py | 85 ++++++++++++++++++++++++++++++++++++++++ 4 files changed, 154 insertions(+), 5 deletions(-) create mode 100644 tools/serve_local.py diff --git a/tools/api_draft.py b/tools/api_draft.py index 412289728..a9e171cc6 100644 --- a/tools/api_draft.py +++ b/tools/api_draft.py @@ -111,11 +111,19 @@ Now write byte-matching C for {t['name']}: - Reply with ONLY one ```c block.""" -# LEAN mode — the SAME prompt shape as tools/format_finetune.py (train/inference must match). +# LEAN mode — keep in sync with tools/format_finetune.py (train/inference must match). The +# "translate EVERY instruction / never-empty" clause was added 2026-06-30 after a prompt test took +# the small-leaf band 0/3 -> 2/3 MATCH (the v2 corpus overfit an empty `void f(void){}` leaf pattern; +# the instruction it most often dropped was the return value / a store). MIRROR this in format_finetune +# before retraining corpus-v3, else train/inference drift. LEAN_SYS = ("You are an expert at MATCHING decompilation for MIPS (PSX, gcc-2.7.2 -O2 -G0 -mips1 -mcpu=3000 " "-msoft-float + maspsx). Given a function's target assembly, output C that the pinned toolchain " "compiles to BYTE-IDENTICAL machine code. The types u8/u16/u32/s8/s16/s32/f32/s64/u64/f64 are " - "predefined (common.h). Output ONLY the C (the function definition + any externs it needs).") + "predefined (common.h). Output ONLY the C (the function definition + any externs it needs). " + "Translate EVERY instruction — NEVER output an empty body. A `jr $ra` with `addiu $v0,$zero,N` " + "in its delay slot is `return N;`; a `sw/sh/sb $aK,off($a0)` is a store " + "`*(T*)((u8*)arg0+off)=argK;` (T=s32/s16/s8); a `lw/lh/lb` is a load. Produce C whose compiled " + "output IS the shown instructions.") # Bridge: real OPEN stubs are splat .s (headers, 3-field comment, spaced operands, resolved jal); the diff --git a/tools/export_pairs.py b/tools/export_pairs.py index 0ff0fbe56..6b306f88a 100644 --- a/tools/export_pairs.py +++ b/tools/export_pairs.py @@ -100,6 +100,34 @@ def extract_defs(src_text): return out +def extract_macro_defs(text): + """corpus-v3: mine the shared `#define DEFINE_func_XXXX() ` macros from engine_core.h. + + These hold the byte-matched SHARED engine functions (setters, return-const, dispatchers) that + extract_defs() can't see — it matches only column-0 inline `func_X(){...}` defs, never the macro + instantiations. The model trained without them (96.6% of corpus-v2 was overlay-unique inline defs), + so it OVERFIT an empty `void f(void){}` leaf pattern and drafts trivial setters/returns empty. + Mining the ~1600 macro bodies feeds exactly the missing variety. Inline `/* asm */` annotations are + stripped (the model's output is clean C; train/inference style must match).""" + out, lines, i = {}, text.split('\n'), 0 + while i < len(lines): + m = re.match(r'\s*#define DEFINE_(func_[0-9A-Fa-f]+)\(\)\s*\\\s*$', lines[i]) + if not m: + i += 1 + continue + fn, i, body = m.group(1), i + 1, [] + while i < len(lines): + ln = lines[i].rstrip() + i += 1 + cont = ln.endswith('\\') + body.append(ln[:-1].rstrip() if cont else ln) + if not cont: + break + c = re.sub(r'/\*.*?\*/', '', '\n'.join(body)) # drop inline asm-annotation comments + out[fn] = '\n'.join(l.rstrip() for l in c.split('\n') if l.strip()) + return out + + def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument('--out', default='datasets/match_pairs') @@ -122,7 +150,15 @@ def main(): region = os.path.relpath(os.path.join(root, f), os.path.join(REPO, 'src'))[:-2] for fn, body in extract_defs(open(os.path.join(root, f)).read()).items(): defs.setdefault(fn, (body, region)) - print(' %d func defs in src' % len(defs), file=sys.stderr) + # corpus-v3: ALSO mine the shared DEFINE_func macro bodies (engine_core.h) — the setters/return-const/ + # dispatchers the model is blind to (extract_defs sees only inline col-0 defs, not macro instantiations). + n_inline = len(defs) + ec = os.path.join(REPO, 'src/shared/engine_core.h') + if os.path.exists(ec): + for fn, body in extract_macro_defs(open(ec).read()).items(): + defs.setdefault(fn, (body, 'shared')) + print(' %d func defs in src (%d inline + %d shared macros)' % ( + len(defs), n_inline, len(defs) - n_inline), file=sys.stderr) pairs, skipped = [], {'no_asm': 0} for fn, (body, region) in sorted(defs.items()): diff --git a/tools/format_finetune.py b/tools/format_finetune.py index d2e14e66a..ad41ca045 100644 --- a/tools/format_finetune.py +++ b/tools/format_finetune.py @@ -27,8 +27,22 @@ _ASF = '-Iinclude -march=r3000 -mtune=r3000 -no-pad-sections -O1 -G0'.split() _TD = re.compile(r'^[ \t]*typedef\b.*\b(u8|u16|u32|u64|s8|s16|s32|s64|f32|f64)[ \t]*;[ \t]*\n', re.M) +_ETYPES = None + + +def _engine_types(): + """src/shared/engine_types.h content (the shared Actor-class struct/union/typedefs), cached.""" + global _ETYPES + if _ETYPES is None: + p = os.path.join(REPO, 'src/shared/engine_types.h') + _ETYPES = (open(p).read() + '\n') if os.path.exists(p) else '' + return _ETYPES + + def compiles(c): - src = '#include "common.h"\n' + _TD.sub('', c) + # corpus-v3: inline the shared struct/typedefs so struct-using macro bodies COMPILE and are KEPT + # (else the filter drops every fn that touches an Actor-class field). common.h provides the scalars. + src = '#include "common.h"\n' + _engine_types() + _TD.sub('', c) wd = os.path.join(REPO, '.run/_ft_cc'); os.makedirs(wd, exist_ok=True) open(os.path.join(wd, 't.c'), 'w').write(src) p = subprocess.run([_CPP] + _CPPF + ['.run/_ft_cc/t.c'], capture_output=True, cwd=REPO) @@ -40,10 +54,16 @@ def compiles(c): p = subprocess.run([_AS] + _ASF + ['-o', '.run/_ft_cc/t.o'], input=p.stdout, capture_output=True, cwd=REPO) return p.returncode == 0 +# MUST match api_draft.LEAN_SYS exactly (train/inference align). The "translate EVERY instruction / +# never-empty" clause (added 2026-06-30) fixes the empty-leaf overfit (prompt test: small-leaf 0/3 -> 2/3). SYS = ("You are an expert at MATCHING decompilation for MIPS (PSX, gcc-2.7.2 -O2 -G0 -mips1 -mcpu=3000 " "-msoft-float + maspsx). Given a function's target assembly, output C that the pinned toolchain " "compiles to BYTE-IDENTICAL machine code. The types u8/u16/u32/s8/s16/s32/f32/s64/u64/f64 are " - "predefined (common.h). Output ONLY the C (the function definition + any externs it needs).") + "predefined (common.h). Output ONLY the C (the function definition + any externs it needs). " + "Translate EVERY instruction — NEVER output an empty body. A `jr $ra` with `addiu $v0,$zero,N` " + "in its delay slot is `return N;`; a `sw/sh/sb $aK,off($a0)` is a store " + "`*(T*)((u8*)arg0+off)=argK;` (T=s32/s16/s8); a `lw/lh/lb` is a load. Produce C whose compiled " + "output IS the shown instructions.") def user_msg(asm): diff --git a/tools/serve_local.py b/tools/serve_local.py new file mode 100644 index 000000000..b82d39e24 --- /dev/null +++ b/tools/serve_local.py @@ -0,0 +1,85 @@ +#!/usr/bin/env python3 +"""serve_local.py — minimal OpenAI-compatible server for the fine-tuned BFM matcher, on the GPU. + +Loads the base Qwen2.5-Coder-7B (4-bit) + a LoRA adapter via Unsloth — the SAME .venv-train stack +that trained it, so no llama.cpp build and no SIGILL-prone prebuilt wheel (this host's CPU has no +AVX-512, which crashes the generic llama-cpp-python CUDA wheels). Serves /v1/chat/completions + +/v1/models so tools/api_draft.py + tools/lora_grind.py hit it UNCHANGED. + + .venv-train/bin/python tools/serve_local.py --adapter models/bfm-match-7b --name bfm-match-7b-v2 --port 1234 + API_BASE=http://127.0.0.1:1234/v1 MODEL=bfm-match-7b-v2 ... tools/lora_grind.py ... + +On-demand only (no daemon); SIGTERM/Ctrl-C to stop. One GPU generation at a time (a lock serializes +— lora_grind drafts sequentially anyway). The requested `model` field is ignored; the loaded adapter +is served (so MODEL can be any non-empty string). +""" +import argparse, json, threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--adapter", default="models/bfm-match-7b") + ap.add_argument("--name", default="bfm-match-local") + ap.add_argument("--port", type=int, default=1234) + ap.add_argument("--host", default="127.0.0.1") + ap.add_argument("--max-seq", type=int, default=4096) + a = ap.parse_args() + + from unsloth import FastLanguageModel + import torch + print(f"[serve] loading {a.adapter} (base+adapter, 4-bit, GPU)...", flush=True) + model, tok = FastLanguageModel.from_pretrained(a.adapter, max_seq_length=a.max_seq, load_in_4bit=True) + FastLanguageModel.for_inference(model) + lock = threading.Lock() + print(f"[serve] model ready: {a.name}", flush=True) + + def generate(messages, max_tokens, temperature): + prompt = tok.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) + inputs = tok(prompt, return_tensors="pt").to("cuda") + do_sample = bool(temperature and temperature > 0) + with lock, torch.no_grad(): + out = model.generate(**inputs, max_new_tokens=max_tokens, do_sample=do_sample, + temperature=(temperature or 1.0), pad_token_id=tok.eos_token_id) + return tok.decode(out[0][inputs["input_ids"].shape[1]:], skip_special_tokens=True) + + class H(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def _json(self, code, obj): + b = json.dumps(obj).encode() + self.send_response(code) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(b))) + self.end_headers() + self.wfile.write(b) + + def do_GET(self): + if self.path.rstrip("/").endswith("/models"): + self._json(200, {"object": "list", "data": [{"id": a.name, "object": "model"}]}) + else: + self._json(404, {"error": "not found"}) + + def do_POST(self): + if not self.path.endswith("/chat/completions"): + self._json(404, {"error": "not found"}) + return + n = int(self.headers.get("Content-Length", 0)) + req = json.loads(self.rfile.read(n) or b"{}") + try: + txt = generate(req.get("messages", []), int(req.get("max_tokens", 1024)), + float(req.get("temperature", 0.2))) + self._json(200, {"id": "cmpl", "object": "chat.completion", "model": a.name, + "choices": [{"index": 0, "finish_reason": "stop", + "message": {"role": "assistant", "content": txt}}]}) + except Exception as e: + self._json(500, {"error": str(e)}) + + srv = ThreadingHTTPServer((a.host, a.port), H) + print(f"[serve] {a.name} serving on http://{a.host}:{a.port}/v1", flush=True) + srv.serve_forever() + + +if __name__ == "__main__": + main()