diff --git a/config/regions.tsv b/config/regions.tsv index 411a937..ca5959c 100644 --- a/config/regions.tsv +++ b/config/regions.tsv @@ -80,6 +80,7 @@ 0x80021FA8 0x80021FF4 src/func_80021FA8.c 0x80021FF4 0x8002203C src/func_80021FF4.c 0x80022A60 0x80022A98 src/func_80022A60.c +0x80022E44 0x80022E78 src/func_80022E44.c 0x80022FB8 0x80022FCC src/func_80022FB8.c 0x80022FCC 0x80022FFC src/func_80022FCC.c 0x80023080 0x800230E4 src/func_80023080.c @@ -205,6 +206,7 @@ 0x80042964 0x800429B0 src/func_80042964.c 0x800429B0 0x800429F0 src/func_800429B0.c 0x80042D64 0x80042D88 src/func_80042D64.c +0x80042D88 0x80042DD4 src/func_80042D88.c 0x80043D8C 0x80043DC4 src/func_80043D8C.c 0x80044F58 0x80044FA4 src/func_80044F58.c gp=-D_80121BFC 0x800450C4 0x80045110 src/func_800450C4.c gp=-D_80121BFC @@ -214,6 +216,7 @@ 0x80045540 0x80045578 src/func_80045540.c 0x80045F1C 0x80045FD8 src/func_80045F1C.c 0x800460AC 0x800460D4 src/func_800460AC.c +0x80047468 0x800474B0 src/func_80047468.c gp=-D_80121BFC 0x800474B0 0x80047508 src/func_800474B0.c 0x80047984 0x80047A14 src/func_80047984.c 0x80048128 0x80048180 src/func_80048128.c gp=-D_80121BFC @@ -290,6 +293,7 @@ 0x80068F40 0x80068F6C src/func_80068F40.c 0x80068F6C 0x80068F98 src/func_80068F6C.c 0x80068F98 0x80068FA8 src/func_80068F98.c +0x800690E4 0x80069128 src/func_800690E4.c 0x80069580 0x800695D8 src/func_80069580.c 0x800697A4 0x800697C4 src/func_800697A4.c 0x800697C4 0x800697FC src/func_800697C4.c @@ -455,6 +459,7 @@ 0x800A6934 0x800A6998 src/func_800A6934.c 0x800A6998 0x800A6A18 src/func_800A6998.c gp=-D_80121B88 0x800A6A18 0x800A6A70 src/func_800A6A18.c +0x800A6B38 0x800A6BA0 src/func_800A6B38.c 0x800A6BEC 0x800A6C34 src/func_800A6BEC.c cc1=-G8 0x800A745C 0x800A74BC src/func_800A745C.c 0x800A74BC 0x800A74D0 src/func_800A74BC.c @@ -582,6 +587,7 @@ 0x800FC2CC 0x800FC2D8 src/func_800FC2CC.c 0x800FE844 0x800FE86C src/func_800FE844.c 0x800FE86C 0x800FE878 src/func_800FE86C.c +0x800FE970 0x800FE9C4 src/func_800FE970.c maspsx=epilogue 0x800FEEB8 0x800FEED0 src/func_800FEEB8.c 0x800FEFFC 0x800FF008 src/func_800FEFFC.c 0x800FF43C 0x800FF47C src/func_800FF43C.c cc1bin=gcc-2.8.1-psx @@ -590,6 +596,7 @@ 0x800FFBEC 0x800FFC3C src/func_800FFBEC.c maspsx=epilogue 0x80100038 0x801000A0 src/func_80100038.c maspsx=epilogue 0x80100318 0x80100334 src/func_80100318.c +0x8010036C 0x801003A4 src/func_8010036C.c maspsx=epilogue 0x80100740 0x801007E0 src/func_80100740.c 0x801008DC 0x80100934 src/func_801008DC.c cc1bin=gcc-2.8.1-psx 0x80100964 0x8010097C src/func_80100964.c diff --git a/src/func_80018284.c b/src/func_80018284.c new file mode 100644 index 0000000..5d8a94c --- /dev/null +++ b/src/func_80018284.c @@ -0,0 +1,54 @@ +/* + * func_80018284 — 80 bytes at 0x80018284..0x800182D4 + * + * A 32-byte block copy: fetch a pointer out of the first argument's +0x10 field + * and copy the 32 bytes it addresses into the third argument. Returns 0, and the + * `move v0,zero` lands in the `jr ra` delay slot. + * + * The observed instructions are: + * lw v0,0x10(a0) p = *(void **)(a0 + 0x10) + * nop load-delay + * lw v1,0(p) / lw a0,4(p) / lw a1,8(p) / lw a3,12(p) + * sw v1,0(a2) / sw a0,4(a2) / sw a1,8(a2) / sw a3,12(a2) + * lw v1,16(p) / lw a0,20(p) / lw a1,24(p) / lw a3,28(p) + * sw v1,16(a2) / sw a0,20(a2) / sw a1,24(a2) / sw a3,28(a2) + * jr ra + * _move v0,zero (delay slot) return 0 + * + * THE LEVER, and it is the same one that has now closed several rows in this + * phase: the copy must be a STRUCT ASSIGNMENT, `*a2 = *p;` over a 32-byte struct, + * NOT sixteen word statements and not two 16-byte assignments written out. cc1 + * then emits the 32 bytes as TWO groups of four loads followed by four stores — + * register pressure caps the in-flight loads at four — which is exactly the + * original. The 8-loads/8-stores shape on its own suggests "eight statements"; + * the tell that it is one assignment is that each group of four STORES follows + * its four loads as a batch instead of pairing load/store (cookbook 76/102). + * + * The register rotation in the emitted copy is cc1's, not the source's: the + * source's struct type has no field names in the output, and reading it as + * eight word stores costs the matching bytes because maspsx then inserts a + * load-delay `nop` after every load. + * + * A SECOND argument exists and is never read. The signature therefore has three + * parameters, with the middle one unused — writing it as `(a0, a2)` two-argument + * form would put the destination in `a1` and mismatch every store. + * + * LIMITS: the struct's size (32 bytes, evidenced by the two 16-byte groups) and + * the meaning of the +0x10 pointer field are hypotheses read off the instruction + * shape; the +0x10 field's type is not otherwise evidenced and no alignment or + * aliasing assumption beyond "a plain 32-byte blob" is needed. Only the compiled + * bytes are evidence. + */ + +struct V8 { + int v[8]; +}; + +int func_80018284(int *a0, int a1, struct V8 *a2) +{ + struct V8 *p = *(struct V8 **)(a0 + 4); + + *a2 = *p; + + return 0; +} diff --git a/src/func_80042CE0.c b/src/func_80042CE0.c new file mode 100644 index 0000000..c77a4ed --- /dev/null +++ b/src/func_80042CE0.c @@ -0,0 +1,72 @@ +/* + * func_80042CE0 — 80 bytes at 0x80042CE0..0x80042D30 + * + * RE-DISPATCH POOL ROW (prio 1, worker D, Phase 12). Matched on the second + * spelling; the first was 2 differing bytes and the difference was the ORDER of + * the two loop initialisations (see 2 below). + * + * Search a two-entry table for a word equal to the argument and return the word + * four bytes BELOW the match; return 0 if no entry matches. + * + * The observed instructions are: + * move a1,zero ; i = 0 <- FIRST + * move v1,zero ; off = 0 + * loop: + * lui at,0x8013 + * addu at,at,v1 + * lw v0,-10092(at) ; *(int *)(0x8012D894 + off) (the key field) + * nop + * bne v0,a0,next + * nop + * lui at,0x8013 + * addu at,at,v1 + * lw v0,-10096(at) ; *(int *)(0x8012D890 + off) (the payload) + * j exit + * nop + * next: + * addiu a1,a1,1 ; i++ + * slti v0,a1,2 + * bnez v0,loop + * addiu v1,v1,1428 ; (delay slot) off += 1428 + * move v0,zero + * exit: + * jr ra / nop + * + * THREE things the bytes fix: + * + * 1. THE TWO LOOPS VARIABLES ARE INITIALISED IN ONE `for` HEAD, `i` BEFORE `off`. + * Writing `off = 0;` as its own statement before the loop emits the two + * `move ..,zero` instructions in the opposite order and costs exactly 2 + * differing bytes at an otherwise identical 80. `for (i = 0, off = 0; ...)` + * fixes it, which is the declaration-vs-assignment-order diagnostic of + * cookbook 100. + * 2. THE INCREMENTS ARE A `for`-STEP COMMA LIST, LEFT TO RIGHT: `i++, off += 1428` + * puts `i++` in the body's tail and `off += 1428` in the LOOP BRANCH'S DELAY + * SLOT. The same two statements in the body produce the reverse order. + * 3. THE STRIDE IS 1428 AND THE FIELDS ARE 4 BYTES APART. The key is at + * `base + off + 4` and the payload at `base + off`, with `base = 0x8012D890` + * (the emitted `-10096(at)` off page 0x8013 is `0x80130000 - 10096`, and the + * key's `-10092` is that plus 4 — cookbook 154's page/immediate arithmetic). + * The address is rematerialised per access as `lui at` + `addu at,at,off`, + * which is what a symbol base plus a running integer offset compiles to. + * + * LIMITS: the two displacements, the stride, the entry count (2, from `slti ... ,2`) + * and the field width (one word each) are read off the bytes. The table is named + * for its base address and its contents and meaning are NOT established; the + * 1428-byte stride similarly. The first field is treated as an `int` key because + * only a word compare is performed, and the payload as an `int` because it is + * returned unchanged. + */ +extern int D_8012D890[]; + +int func_80042CE0(int a0) +{ + int i; + int off; + + for (i = 0, off = 0; i < 2; i++, off += 1428) { + if (*(int *)((char *)D_8012D890 + off + 4) == a0) + return *(int *)((char *)D_8012D890 + off); + } + return 0; +} diff --git a/src/func_80042D88.c b/src/func_80042D88.c new file mode 100644 index 0000000..5b1b0f9 --- /dev/null +++ b/src/func_80042D88.c @@ -0,0 +1,68 @@ +/* + * func_80042D88 — 76 bytes at 0x80042D88..0x80042DD4 + * + * Hypothesis, not a claim about meaning: a backwards word-loop over an array whose length is + * derived from a parameter. For each word, from the top down to index 0, it either clears the + * destination word (when the source pointer parameter is NULL) or ANDs the destination with the + * complement of the source word. + * + * Original words: + * 00063143 sra a2,a2,0x5 ; len = n >> 5 (in WORDS) + * 00063080 sll a2,a2,0x2 ; ... and the compiler scales it for the walk + * 00C43821 addu a3,a2,a0 ; INDEX FIRST: dp = &d[len] + * 00C53021 addu a2,a2,a1 ; INDEX FIRST: sp = &s[len] + * 2484FFFC addiu a0,a0,-4 ; end = &d[-1] + * $L: 14A00003 bnez a1,0x80042DAC ; s != 0 -> the complement arm (branch INTO the body) + * 00000000 nop + * 08010B70 j 0x80042DC0 ; the NULL arm jumps over the complement block + * ACE00000 sw zero,0(a3) ; (delay) *dp = 0 + * $B: 8CC20000 lw v0,0(a2) ; (L) v0 = *sp + * 8CE30000 lw v1,0(a3) ; (L) v1 = *dp + * 00021027 nor v0,zero,v0 ; v0 = ~v0 + * 00621824 and v1,v1,v0 ; v1 &= v0 + * ACE30000 sw v1,0(a3) ; *dp = v1 + * $T: 24E7FFFC addiu a3,a3,-4 ; dp-- + * 14E4FFF5 bne a3,a0,0x80042D9C ; dp != end -> loop + * 24C6FFFC addiu a2,a2,-4 ; (delay) sp-- + * 03E00008 jr ra + * 00000000 nop + * + * FOUR SPELLINGS (80 / 76 / 76 / 76, the last three differing only in the two `addu`s and the + * loop test), AND THE CLASS WAS "strength-reduce (sra5+sll2 vs cc1 folded shift)": + * + * (1) **THE LOOP MUST BE AN INDEXED do/while TERMINATING ON `i != -1`.** This is what makes cc1 + * strength-reduce the array walk into a POINTER walk (`bne a3,a0`) with NO counter register + * and `end = &d[-1]`. A `for (i = len; i >= 0; i--)` keeps a counter AND emits a redundant + * entry guard (`bltz`, +4 bytes); a hand-written pointer walk is `addu ,,` + * with the BASE first. + * (2) **THE INDEXED FORM IS ALSO WHAT SETS THE COMMUTATIVE `addu` ORDER.** The original's two + * pointer set-ups are INDEX-FIRST (`addu a3,a2,a0`), which cc1 produces from the + * strength-reduced indexed loop; writing the walk by hand with `p = (char *)d + len` gives + * BASE-first and 3 differing bytes (cookbook 22/103: the spelling decides, and here the + * deciding spelling is "let the compiler do the strength reduction"). + * (3) **THE NULL TEST MUST BE WRITTEN INVERTED WITH THE ARMS SWAPPED.** `if (s == 0) *dp = 0; + * else *dp = *dp & ~*sp;` reproduces the original's `bnez` that branches INTO the complement + * body with the NULL store in the following `j`'s delay slot; the natural `if (s != 0) ...` + * mirrors the layout (cookbook 28/77/100). + * + * LIMITS: the function name, the shift 5, the parameter roles, the complement operation and the + * loop direction are read from the instruction encodings; only the bytes are evidence. `len` is + * a WORD count because the emitted `sll 2` scales it for the walk, and the `sra` makes `n` a + * SIGNED value. The parameters' types beyond `int *`/`int` and the array's meaning are + * unestablished; the NULL test on `s` is a condition on the PARAMETER, not on the walked pointer + * (a2 is never compared against zero), which is why it is written `if (s == 0)`. + */ + +void func_80042D88(int *d, int *s, int n) +{ + int len = n >> 5; + int i = len; + + do { + if (s == 0) + d[i] = 0; + else + d[i] = d[i] & ~s[i]; + i--; + } while (i != -1); +} diff --git a/src/func_80094370.c b/src/func_80094370.c new file mode 100644 index 0000000..ea7cae5 --- /dev/null +++ b/src/func_80094370.c @@ -0,0 +1,59 @@ +/* + * func_80094370 — 84 bytes at 0x80094370..0x800943C4 + * + * Copies a 72-byte object into the object reached by a triple indirection from + * argument 0, while the second argument is the copy's SOURCE. + * + * The observed instructions are: + * lw v0,0xC(a0) \ v0 = *(void **)(a0 + 0xC) + * move a3,a1 / a3 = the copy SOURCE (argument 1) + * lw v0,0x160(v0) v0 = *(void **)(v0 + 0x160) (third level) + * addiu t0,a3,0x40 t0 = source + 0x40 (the loop end, precomputed) + * addiu a2,v0,0x160 a2 = DESTINATION + * loop: + * lw v0,0(a3) / lw v1,4(a3) / lw a0,8(a3) / lw a1,12(a3) + * sw v0,0(a2) / sw v1,4(a2) / sw a0,8(a2) / sw a1,12(a2) + * addiu a3,a3,0x10 \ + * bne t0,a3,loop / 16 bytes per iteration, 4 iterations + * _addiu a2,a2,0x10 (delay slot) + * lw v0,0(a3) \ the 8-byte TAIL + * lw v1,4(a3) | + * sw v0,0(a2) | + * jr ra | + * _sw v1,4(a2) / (in the delay slot) + * + * 72 bytes = 4 x 16 + 8, which is exactly how cc1 expands a large struct + * assignment: a 16-byte-per-iteration loop with a trailing partial block. So the + * whole copy is ONE statement, `*dst = *src;` over a 72-byte struct — writing the + * loop by hand, or the 18 word stores out, gives a different shape. This is the + * same lever as cookbook 76/102 (the struct assignment is what batches the loads), + * here large enough that cc1 chooses a loop instead of an unrolled run. + * + * THE ONE-BYTE TRAP, and it is worth stating because the row is otherwise exact: + * the copy source is **argument 1**, not argument 0 — `move a3,a1`. Reading the + * disassembly as `move a3,a0` (argument 0) gives a candidate that differs by + * exactly ONE BYTE, in the `rs` field of that move, and the whole 84-byte row then + * reports differing_bytes=1. Argument 0 is used only for the dereference chain. + * + * The destination is `*(void **)(*(void **)(a0 + 0xC) + 0x160) + 0x160`: one load + * off a0, one load off that result, then a BYTE offset of 0x160. The end pointer + * (`source + 0x40`) is computed before the loop, so the loop compares pointers + * rather than counting — cc1's loop optimisation produced that from a counted + * loop, and it is not evidence about the source's spelling. + * + * LIMITS: the struct's size (72 bytes, evidenced by 4x16 + 8), the field offsets + * (0xC, 0x160, 0x160), the pointer types and the fact that the source object's + * trailing 8 bytes are copied separately are all read off the instruction shape. + * Only the compiled bytes are evidence. + */ + +struct Big { + int v[18]; +}; + +void func_80094370(int *a0, struct Big *a1) +{ + struct Big *d = (struct Big *)((char *)*(int **)((char *)*(int **)((char *)a0 + 0xC) + 0x160) + 0x160); + + *d = *a1; +} diff --git a/src/func_80099D14.c b/src/func_80099D14.c new file mode 100644 index 0000000..d0a39f2 --- /dev/null +++ b/src/func_80099D14.c @@ -0,0 +1,33 @@ +/* func_80099D14 — 88 bytes at 0x80099D14..0x80099D6C [DRAFT] */ +extern int D_80139C30; + +struct func_80099D14_S { + int f0; + unsigned char f4; + unsigned char f5; + unsigned char f6; + unsigned char f7; + unsigned char f8; + unsigned char pad[3]; + int f12; + int f16; + int f20; +}; + +void func_80099AE4(int *, int *); + +int func_80099D14(int a0, unsigned char a1, int a2, unsigned char a3) +{ + struct func_80099D14_S s; + + s.f0 = a0; + s.f4 = 2; + s.f5 = a1; + s.f6 = (a3 != 0) << 1; + s.f7 = 0; + s.f8 = 0; + s.f12 = a2; + s.f16 = 0; + func_80099AE4((int *)&s, &D_80139C30); + return 1; +} diff --git a/src/func_800B5C5C.c b/src/func_800B5C5C.c new file mode 100644 index 0000000..1a90a12 --- /dev/null +++ b/src/func_800B5C5C.c @@ -0,0 +1,18 @@ +/* func_800B5C5C — 88 bytes at 0x800B5C5C..0x800B5CB4 [DRAFT] */ +extern int D_80122068; +extern int D_8012277C; +extern unsigned short D_80122738; + +void func_800B5C5C(int a0, int a1) +{ + int i = D_80122068; + int prev = D_8012277C; + unsigned short s = D_80122738; + + D_8012277C = a0; + *(short *)((char *)0x80140480 + i * 2) = s; + D_80122068 = i + 1; + *(int *)((char *)0x80140930 + i * 4) = prev; + if ((a1 & 0xffff) != 11) + D_80122738 = a1; +} diff --git a/tools/sf3_free b/tools/sf3_free index 36663d4..659e2d6 100755 --- a/tools/sf3_free +++ b/tools/sf3_free @@ -10,8 +10,8 @@ Two checks, in order: the region is `[A, B)`, so **B itself is NOT in it**. A naive `grep` reports a free address as taken whenever it happens to be the exclusive end of the preceding region. 2. **.run/pN/inflight.tsv** -- the write-ahead ledger, read with **LAST-ROW-WINS**. The - effective state of an address is its *last* row: `wip` means held, `released`/`claimed` - mean free. + effective state of an address is its *last* row: `wip` and `claimed` mean held, + `released` means free. `claimed` is HELD, and that is a correction -- see below. *** THE LEDGER IS PER-PHASE, AND ONE PHASE'S LEDGER IS DEAD TO THE NEXT *** @@ -188,6 +188,17 @@ def main(argv: list[str]) -> int: print(f"{text} FREE (no ledger row)") elif state[0] == "wip": print(f"{text} TAKEN by inflight: {state[1]}") + elif state[0] == "claimed": + # HELD, not free. A `claimed` row means a worker has STAGED a verified claim and is + # waiting for the orchestrator to merge it. Treating that window as free is a false + # FREE in the dangerous direction: the row is neither registered nor abandoned, so the + # next worker re-derives it and the merge can then receive two claims for one address. + # Worker D found this and reported it rather than changing a tracked tool itself. + # The window is not hypothetical -- it is exactly what the credit outage produced, when + # all four workers stopped with claims in flight and the pools were re-dispatched. + # Note a registered row never reaches here: region_owner() above already returns TAKEN, + # so this branch only ever fires for a claim whose merge has NOT landed. + print(f"{text} TAKEN by inflight (claim staged, merge pending): {state[1]}") else: print(f"{text} FREE (last ledger row '{state[0]}'): {state[1]}") return 0