From 8f37e15f8ad3f504e0efe7e5032efcb9327f4291 Mon Sep 17 00:00:00 2001 From: rajames Date: Fri, 2 Oct 2026 23:33:57 -0400 Subject: [PATCH] feat(v4.0.0): repack Q.* so no UM* call holds all four inputs Q.* now forms its products in the order a1*b1 (low cell), a0*b1, a1*b0, a0*b0, dropping each input after its last use, with b0 waiting on the return stack. Each sign correction, (b1<0 ? a0 : 0) and (a1<0 ? b0 : 0), is folded into cell 2 as soon as its operands are adjacent, using -if rather than 0< calls. At most four live values sit under any UM* call. Headroom (data cells under args / return entries): 2/2 -> 3/3. Same checks as before at 32- and 64-bit cells, optimised and ASan+UBSan; dropping either sign correction fails about 9900 checks. Co-Authored-By: Claude Opus 5.5 --- docs/v4.0.0/DECOMPOSITION.md | 2 +- v4/tests/test_foundation.c | 72 ++++++++++++++++++++++-------------- 2 files changed, 45 insertions(+), 29 deletions(-) diff --git a/docs/v4.0.0/DECOMPOSITION.md b/docs/v4.0.0/DECOMPOSITION.md index 835f669c..1a160c5f 100644 --- a/docs/v4.0.0/DECOMPOSITION.md +++ b/docs/v4.0.0/DECOMPOSITION.md @@ -656,7 +656,7 @@ word. Per D-10 it occupies two cells on a 64-bit node too. Per D-8, v4 Q values | Word | Fate | Notes | | --- | --- | --- | | `Q.+` `Q.-` | CAP | `D+`, `D-` — executed on the golden model (2026-10-02) against v3's `q48_add`/`q48_sub` (`uint64_t` wrapping): bit-for-bit at 32-bit cells; at 64-bit cells the low cell is v3's, per D-10. | -| `Q.*` | CAP | `2OVER drop over 0< over and push over UM* pop - push push push over pop UM* drop pop pop push SWAP pop + push push over 0< over and pop pop push SWAP pop SWAP - push push over over UM* pop pop D+ push push push drop pop UM* 0 pop pop D+ push dup push Q.TO-INT pop pop Q.TO-INT` — `floor(a*b / 2^16)`, signed (D-8), cut to two cells. Cells 0–2 of the product come from the four unsigned cell products `a0*b0`, `a0*b1`, `a1*b0`, `a1*b1` (low cell only); reading `a1` and `b1` as signed takes `(a1<0 ? b0 : 0) + (b1<0 ? a0 : 0)` off cell 2; the result is cells 0–2 shifted right 16 with `Q.TO-INT` twice. Executed on the golden model (2026-10-02) against an independent limb-by-limb reference, and against v3's `q48_mul` for non-negative operands. **Stack use is tight:** it leaves its caller 2 data cells under its arguments and 2 return entries (1 before `UM*` was revised); it holds all four input cells across its `UM*` calls. To be improved. Clobbers `A`. | +| `Q.*` | CAP | `SWAP push over over UM* drop push push over pop -if L1 SWAP jump L2 L1: SWAP drop 0 L2: pop SWAP - push push over pop UM* pop + ROT pop dup push over -if L3 drop dup jump L4 L3: drop 0 L4: push UM* pop - D+ ROT pop UM* SWAP push 0 D+ pop push over pop SWAP Q.TO-INT push Q.TO-INT pop SWAP` (`SWAP` and `ROT` in line) — `floor(a*b / 2^16)`, signed (D-8), cut to two cells. Cells 0–2 of the product come from the unsigned cell products `a1*b1` (low cell), `a0*b1`, `a1*b0`, `a0*b0`, in that order, each input dropped after its last use and `b0` waiting on the return stack. Reading `a1` and `b1` as signed takes `(b1<0 ? a0 : 0)` and `(a1<0 ? b0 : 0)` off cell 2; each is folded in as soon as its operands are adjacent. The result is cells 0–2 shifted right 16 with `Q.TO-INT` twice. Executed on the golden model (2026-10-02) against an independent limb-by-limb reference, and against v3's `q48_mul` for non-negative operands. Leaves its caller 3 data cells under its arguments and 3 return entries. Clobbers `A`. | | `Q./` | CAP | Shifted long division. **Division by zero sets an error** (v3 returned 0 silently). | | `Q.ABS` `Q.NEG` | CAP | `DABS`, `DNEGATE` — executed on the golden model (2026-10-02) against v3's `q48_abs` and `0 - q`: bit-for-bit at 32-bit cells, including v3's wrap of Q min to itself; at 64-bit cells the low cell is v3's and the high cell is the true sign, per D-10. | | `Q.LOG` `Q.EXP` `Q.SQRT` `Q.SIN` `Q.COS` | CAP | Algorithms ported from `q48_words.c`; the hosted C versions are the golden model. | diff --git a/v4/tests/test_foundation.c b/v4/tests/test_foundation.c index 46171992..1bcabfa7 100644 --- a/v4/tests/test_foundation.c +++ b/v4/tests/test_foundation.c @@ -548,35 +548,51 @@ static void build(void) * shifted right 16. The cell products are unsigned; reading a1 and b1 * as signed takes (a1<0 ? b0 : 0) + (b1<0 ? a0 : 0) off cell 2, the * only cell above the unsigned product's that the result reaches. - * 2OVER drop over 0< over and push R: b1<0 ? a0 : 0 - * over UM* pop - a0 a1 b0 b1 l01 h01 a0*b1, corrected - * push push R: h01 l01 - * push over pop UM* drop a0 a1 b0 t3 a1*b1 low - * pop pop push SWAP pop + a0 a1 b0 x0 x1 - * push push over 0< over and a0 a1 b0 c R: x1 x0 - * pop pop push SWAP pop SWAP - a0 a1 b0 x0 x1-c - * push push over over UM* a0 a1 b0 l10 h10 a1*b0 - * pop pop D+ a0 a1 b0 X0 X1 - * push push push drop pop a0 b0 R: X1 X0 - * UM* 0 pop pop D+ c0 c1 c2 - * push dup push Q.TO-INT pop pop Q.TO-INT ; - * SWAP in line. Clobbers A. */ + * Products in the order a1*b1 (low cell), a0*b1, a1*b0, a0*b0, each + * input dropped after its last use; b0 waits on the return stack. + * SWAP push over over UM* drop a0 a1 b1 t3 R: b0 + * push push over pop a0 a1 a0 b1 R: b0 t3 + * -if L1 SWAP jump L2 L1: SWAP drop 0 L2: a0 a1 b1 c2 + * pop SWAP - push a0 a1 b1 R: b0 t3-c2 + * push over pop UM* pop + a0 a1 x0 x1 R: b0 + * ROT pop dup push a0 x0 x1 a1 b0 + * over -if L3 drop dup jump L4 L3: drop 0 L4: ... a1 b0 c1 + * push UM* pop - D+ a0 X0 X1 + * ROT pop UM* X0 X1 l00 h00 + * SWAP push 0 D+ pop c1 c2 c0 + * push over pop SWAP Q.TO-INT push Q.TO-INT pop SWAP ; + * SWAP and ROT in line. Clobbers A. */ #define SWAP_INLINE() do { O(OVER); O(PUSH); O(PUSH); O(DROP); O(RPOP); O(RPOP); } while (0) - w_qstar = v4_asm_label(&as); - CALL(w_2over); O(DROP); - O(OVER); CALL(w_zless); O(OVER); O(AND); O(PUSH); - O(OVER); CALL(w_umstar); O(RPOP); CALL(w_minus); - O(PUSH); O(PUSH); - O(PUSH); O(OVER); O(RPOP); CALL(w_umstar); O(DROP); - O(RPOP); O(RPOP); O(PUSH); SWAP_INLINE(); O(RPOP); O(ADD); - O(PUSH); O(PUSH); O(OVER); CALL(w_zless); O(OVER); O(AND); - O(RPOP); O(RPOP); O(PUSH); SWAP_INLINE(); O(RPOP); SWAP_INLINE(); CALL(w_minus); - O(PUSH); O(PUSH); O(OVER); O(OVER); CALL(w_umstar); - O(RPOP); O(RPOP); CALL(w_dplus); - O(PUSH); O(PUSH); O(PUSH); O(DROP); O(RPOP); - CALL(w_umstar); LIT(0); O(RPOP); O(RPOP); CALL(w_dplus); - O(PUSH); O(DUP); O(PUSH); CALL(w_qtoint); O(RPOP); O(RPOP); CALL(w_qtoint); - O(SEMI); +#define ROT_INLINE() do { O(PUSH); SWAP_INLINE(); O(RPOP); SWAP_INLINE(); } while (0) + { + v4_asm_ref l1, l2, l3, l4; + w_qstar = v4_asm_label(&as); + SWAP_INLINE(); O(PUSH); O(OVER); O(OVER); CALL(w_umstar); O(DROP); + O(PUSH); O(PUSH); O(OVER); O(RPOP); + l1 = v4_asm_branch_fwd(&as, V4_OP_MINUS_IF); + SWAP_INLINE(); + l2 = v4_asm_branch_fwd(&as, V4_OP_JUMP); + v4_asm_resolve(&as, l1, v4_asm_label(&as)); + SWAP_INLINE(); O(DROP); LIT(0); + v4_asm_resolve(&as, l2, v4_asm_label(&as)); + O(RPOP); SWAP_INLINE(); CALL(w_minus); O(PUSH); + O(PUSH); O(OVER); O(RPOP); CALL(w_umstar); O(RPOP); O(ADD); + ROT_INLINE(); O(RPOP); O(DUP); O(PUSH); + O(OVER); + l3 = v4_asm_branch_fwd(&as, V4_OP_MINUS_IF); + O(DROP); O(DUP); + l4 = v4_asm_branch_fwd(&as, V4_OP_JUMP); + v4_asm_resolve(&as, l3, v4_asm_label(&as)); + O(DROP); LIT(0); + v4_asm_resolve(&as, l4, v4_asm_label(&as)); + O(PUSH); CALL(w_umstar); O(RPOP); CALL(w_minus); CALL(w_dplus); + ROT_INLINE(); O(RPOP); CALL(w_umstar); + SWAP_INLINE(); O(PUSH); LIT(0); CALL(w_dplus); O(RPOP); + O(PUSH); O(OVER); O(RPOP); SWAP_INLINE(); CALL(w_qtoint); + O(PUSH); CALL(w_qtoint); O(RPOP); SWAP_INLINE(); + O(SEMI); + } +#undef ROT_INLINE #undef SWAP_INLINE CHECK(v4_asm_ok(&as), "foundation words assemble");