diff --git a/Makefile b/Makefile index dbd98669..14d356b0 100644 --- a/Makefile +++ b/Makefile @@ -231,6 +231,7 @@ TESTS = $(BIN)/test_smoke $(BIN)/test_lex $(BIN)/test_parse $(BIN)/test_check \ $(BIN)/test_dot_explicit_deref \ $(BIN)/test_nested_structlit \ $(BIN)/test_dot_structlit \ + $(BIN)/test_nested_call_rhs \ $(BIN)/test_use_promote_alias \ $(BIN)/test_field_signed $(BIN)/test_frame_argcount \ $(BIN)/test_selfhost $(BIN)/test_w6a_ww $(BIN)/test_w6l_ww \ @@ -402,6 +403,12 @@ $(BIN)/test_dot_structlit: test/wcc/704_dot_structlit.c \ $(LIB)/libwwrt.a | $(BIN) $(CC) $(CFLAGS) -o $@ $< +$(BIN)/test_nested_call_rhs: test/wcc/705_nested_call_rhs.c \ + $(BIN)/ww $(BIN)/w6c $(BIN)/w6a $(BIN)/w6l \ + $(BIN)/ww_ww $(BIN)/w6c_ww $(BIN)/w6a_ww $(BIN)/w6l_ww \ + $(LIB)/libwwrt.a | $(BIN) + $(CC) $(CFLAGS) -o $@ $< + $(BIN)/test_use_promote_alias: test/wcc/699_use_promote_alias.c \ $(BIN)/ww $(BIN)/w6c $(BIN)/w6a $(BIN)/w6l \ $(LIB)/libwwrt.a | $(BIN) diff --git a/cmd/w6c/cgen.c b/cmd/w6c/cgen.c index a4098133..662eb865 100644 --- a/cmd/w6c/cgen.c +++ b/cmd/w6c/cgen.c @@ -1347,6 +1347,55 @@ cg_structlit_fill(Cg *c, Local **locals_p, Type *lu, Node *lit, mode, srcoff, name, disp + (int)foff); continue; } + /* Nested struct-typed CALL value (#20). cgexpr leaves + * AX=bytes[0..7], DX=bytes[8..15], CX=bytes[16..23] per + * #4's cgreturn ABI. Pre-#20 the cgexpr-then-AX-store + * fallthrough below silently dropped past the first + * qword for any fsz > 8 (only AX got stored). + * + * Sized stores: MOVQ for full 8B chunks plus a sized tail + * (MOVL/MOVW/MOVB) by `tail = fsz%8`. Mirrors #4's receive + * shape at the N_LET / N_ASSIGN call-rhs sites; the + * MOVW-for-tail==2 emission only fires on shapes that + * didn't compile before, so no #13 byte-identity concern. + * + * Guard `fsz <= 24 && fsz%8 ∈ {0,1,2,4}` matches #4's + * cgreturn ABI: >24B falls through (sret deferred); + * fsz%8 ∈ {3,5,6,7} would need shift-store and is also + * unsupported by #4 — falls through to the existing + * AX-only wrongness (consistent, tracked as follow-up). + * + * INVARIANT: between cgexpr(N_CALL) and the AX/DX/CX + * stores below, NO instruction may touch AX/DX/CX. The + * BX reload (MOVQ/LEAQ) is safe; any other emission + * added here will silently corrupt the return value. */ + if (fu && fu->kind == TY_STRUCT + && f->lhs && f->lhs->kind == N_CALL + && fsz <= 24 + && (fsz % 8 == 0 || fsz % 8 == 1 + || fsz % 8 == 2 || fsz % 8 == 4)) { + cgexpr(c, f->lhs, *locals_p); + if (mode == DST_PTR_LOCAL) + ins2(c, A_MOVQ, amem(D_BP, srcoff), + areg(D_BX)); + else if (mode == DST_GLOBAL) + ins2(c, A_LEAQ, masym(c, name), areg(D_BX)); + int regs[3] = { D_AX, D_DX, D_CX }; + int full = fsz / 8; + int tail = fsz % 8; + for (int i = 0; i < full; i++) + ins2(c, A_MOVQ, areg(regs[i]), + amem(base_reg, + disp + (int)foff + i * 8)); + if (tail > 0) { + int op = (tail == 4) ? A_MOVL + : (tail == 2) ? A_MOVW : A_MOVB; + ins2(c, op, areg(regs[full]), + amem(base_reg, + disp + (int)foff + full * 8)); + } + continue; + } cgexpr(c, f->lhs, *locals_p); /* For non-BP modes, cgexpr just clobbered BX; reload it * before the store. */ diff --git a/selfhost/cmd/w6c/main.combined.ww b/selfhost/cmd/w6c/main.combined.ww index 41ea38ea..3dcf8f77 100644 --- a/selfhost/cmd/w6c/main.combined.ww +++ b/selfhost/cmd/w6c/main.combined.ww @@ -8826,8 +8826,123 @@ fn cgstructlitfill(c: *cgen, si: *structinfo, lit: *node, }; }; }; + // Nested struct-typed CALL value (#20). cgexpr + // leaves AX=bytes[0..7], DX=bytes[8..15], CX= + // bytes[16..23] per #4's cgreturn ABI. Pre-#20 + // the cgexpr-then-AX-store fallthrough below + // silently dropped past the first qword for any + // fsz > 8 (only AX got stored). + // + // Sized stores: MOVQ for full 8B chunks plus a + // sized tail (MOVL/MOVW/MOVB) by `tail = fsz%8`. + // Mirror of cstage cg_structlit_fill's #20 branch. + // MOVW-for-tail==2 only fires on shapes that + // didn't compile before, so no #13 byte-identity + // concern. + // + // Guard `fsz <= 24 && fsz%8 ∈ {0,1,2,4}` matches + // #4's cgreturn ABI: >24B falls through (sret + // deferred); fsz%8 ∈ {3,5,6,7} would need shift- + // store and is also unsupported by #4 — falls + // through to the existing AX-only wrongness + // (consistent, tracked as follow-up). + // + // INVARIANT: between cgexpr(N_CALL) and the + // AX/DX/CX stores below, NO instruction may touch + // AX/DX/CX. The BX reload is safe; any other + // emission added here will silently corrupt the + // return value. + let callwhole: bool = false; + if (!nested) { + if (fieldnode.lhs != nil) { + if (fieldnode.lhs.kind == nkind.N_CALL) { + if (fi.tnode != nil) { + if (fi.tnode.kind == nkind.N_TNAME) { + if (primsize(fi.tnode.str) == 0) { + let csi: *structinfo = structlookup(c, fi.tnode.str); + if (csi != nil) { + // Use the inner struct's + // NATURAL size (no 8B slot + // rounding) so MOVL/MOVW/ + // MOVB tail dispatch matches + // cstage's fl->type->size + // (which is natural per + // check.c). fi.fsz here is + // wwstage's slot-padded + // totsize — using it would + // emit 2× MOVQ where cstage + // emits MOVQ+MOVL for a + // 12B inner, etc. (task #15 + // territory; sidestepped + // locally.) + let cfsz: i32 = structnaturalsize(csi); + let crem: i32 = cfsz - (cfsz / 8) * 8; + if (cfsz <= 24) { + if (crem == 0 || crem == 1 + || crem == 2 || crem == 4) { + cgexpr(c, fieldnode.lhs); + if (mode == 1) { + emitline("\tMOVQ\t"); + emitoff(srcoff: i64); + emitline("(BP), BX\n"); + }; + if (mode == 2) { + emitline("\tLEAQ\t"); + emitsymname(c, srcname); + emitline("(SB), BX\n"); + }; + let full: i32 = cfsz / 8; + let ci: i32 = 0; + for (ci < full) { + let r: str = "AX"; + if (ci == 1) { r = "DX"; }; + if (ci == 2) { r = "CX"; }; + emitline("\tMOVQ\t"); + emitline(r); + emitline(", "); + if (mode == 0) { + emitoff((disp + fi.foff + ci * 8): i64); + emitline("(BP)\n"); + } else { + emitdispreg((disp + fi.foff + ci * 8): i64, basereg); + emitline("\n"); + }; + ci += 1; + }; + if (crem > 0) { + let top: str = "MOVB"; + if (crem == 4) { top = "MOVL"; }; + if (crem == 2) { top = "MOVW"; }; + let tr: str = "AX"; + if (full == 1) { tr = "DX"; }; + if (full == 2) { tr = "CX"; }; + emitline("\t"); + emitline(top); + emitline("\t"); + emitline(tr); + emitline(", "); + if (mode == 0) { + emitoff((disp + fi.foff + full * 8): i64); + emitline("(BP)\n"); + } else { + emitdispreg((disp + fi.foff + full * 8): i64, basereg); + emitline("\n"); + }; + }; + callwhole = true; + }; + }; + }; + }; + }; + }; + }; + }; + }; if (nested) { fi = nil; + } else if (callwhole) { + fi = nil; } else { cgexpr(c, fieldnode.lhs); // For non-BP modes, cgexpr just clobbered diff --git a/selfhost/cmd/wcc/cgenutil.ww b/selfhost/cmd/wcc/cgenutil.ww index b7eca4d0..fb103a93 100644 --- a/selfhost/cmd/wcc/cgenutil.ww +++ b/selfhost/cmd/wcc/cgenutil.ww @@ -2954,8 +2954,123 @@ fn cgstructlitfill(c: *cgen, si: *structinfo, lit: *node, }; }; }; + // Nested struct-typed CALL value (#20). cgexpr + // leaves AX=bytes[0..7], DX=bytes[8..15], CX= + // bytes[16..23] per #4's cgreturn ABI. Pre-#20 + // the cgexpr-then-AX-store fallthrough below + // silently dropped past the first qword for any + // fsz > 8 (only AX got stored). + // + // Sized stores: MOVQ for full 8B chunks plus a + // sized tail (MOVL/MOVW/MOVB) by `tail = fsz%8`. + // Mirror of cstage cg_structlit_fill's #20 branch. + // MOVW-for-tail==2 only fires on shapes that + // didn't compile before, so no #13 byte-identity + // concern. + // + // Guard `fsz <= 24 && fsz%8 ∈ {0,1,2,4}` matches + // #4's cgreturn ABI: >24B falls through (sret + // deferred); fsz%8 ∈ {3,5,6,7} would need shift- + // store and is also unsupported by #4 — falls + // through to the existing AX-only wrongness + // (consistent, tracked as follow-up). + // + // INVARIANT: between cgexpr(N_CALL) and the + // AX/DX/CX stores below, NO instruction may touch + // AX/DX/CX. The BX reload is safe; any other + // emission added here will silently corrupt the + // return value. + let callwhole: bool = false; + if (!nested) { + if (fieldnode.lhs != nil) { + if (fieldnode.lhs.kind == nkind.N_CALL) { + if (fi.tnode != nil) { + if (fi.tnode.kind == nkind.N_TNAME) { + if (primsize(fi.tnode.str) == 0) { + let csi: *structinfo = structlookup(c, fi.tnode.str); + if (csi != nil) { + // Use the inner struct's + // NATURAL size (no 8B slot + // rounding) so MOVL/MOVW/ + // MOVB tail dispatch matches + // cstage's fl->type->size + // (which is natural per + // check.c). fi.fsz here is + // wwstage's slot-padded + // totsize — using it would + // emit 2× MOVQ where cstage + // emits MOVQ+MOVL for a + // 12B inner, etc. (task #15 + // territory; sidestepped + // locally.) + let cfsz: i32 = structnaturalsize(csi); + let crem: i32 = cfsz - (cfsz / 8) * 8; + if (cfsz <= 24) { + if (crem == 0 || crem == 1 + || crem == 2 || crem == 4) { + cgexpr(c, fieldnode.lhs); + if (mode == 1) { + emitline("\tMOVQ\t"); + emitoff(srcoff: i64); + emitline("(BP), BX\n"); + }; + if (mode == 2) { + emitline("\tLEAQ\t"); + emitsymname(c, srcname); + emitline("(SB), BX\n"); + }; + let full: i32 = cfsz / 8; + let ci: i32 = 0; + for (ci < full) { + let r: str = "AX"; + if (ci == 1) { r = "DX"; }; + if (ci == 2) { r = "CX"; }; + emitline("\tMOVQ\t"); + emitline(r); + emitline(", "); + if (mode == 0) { + emitoff((disp + fi.foff + ci * 8): i64); + emitline("(BP)\n"); + } else { + emitdispreg((disp + fi.foff + ci * 8): i64, basereg); + emitline("\n"); + }; + ci += 1; + }; + if (crem > 0) { + let top: str = "MOVB"; + if (crem == 4) { top = "MOVL"; }; + if (crem == 2) { top = "MOVW"; }; + let tr: str = "AX"; + if (full == 1) { tr = "DX"; }; + if (full == 2) { tr = "CX"; }; + emitline("\t"); + emitline(top); + emitline("\t"); + emitline(tr); + emitline(", "); + if (mode == 0) { + emitoff((disp + fi.foff + full * 8): i64); + emitline("(BP)\n"); + } else { + emitdispreg((disp + fi.foff + full * 8): i64, basereg); + emitline("\n"); + }; + }; + callwhole = true; + }; + }; + }; + }; + }; + }; + }; + }; + }; if (nested) { fi = nil; + } else if (callwhole) { + fi = nil; } else { cgexpr(c, fieldnode.lhs); // For non-BP modes, cgexpr just clobbered diff --git a/selfhost/cmd/wwdump/main.combined.ww b/selfhost/cmd/wwdump/main.combined.ww index 6b416dfa..d0897870 100644 --- a/selfhost/cmd/wwdump/main.combined.ww +++ b/selfhost/cmd/wwdump/main.combined.ww @@ -8826,8 +8826,123 @@ fn cgstructlitfill(c: *cgen, si: *structinfo, lit: *node, }; }; }; + // Nested struct-typed CALL value (#20). cgexpr + // leaves AX=bytes[0..7], DX=bytes[8..15], CX= + // bytes[16..23] per #4's cgreturn ABI. Pre-#20 + // the cgexpr-then-AX-store fallthrough below + // silently dropped past the first qword for any + // fsz > 8 (only AX got stored). + // + // Sized stores: MOVQ for full 8B chunks plus a + // sized tail (MOVL/MOVW/MOVB) by `tail = fsz%8`. + // Mirror of cstage cg_structlit_fill's #20 branch. + // MOVW-for-tail==2 only fires on shapes that + // didn't compile before, so no #13 byte-identity + // concern. + // + // Guard `fsz <= 24 && fsz%8 ∈ {0,1,2,4}` matches + // #4's cgreturn ABI: >24B falls through (sret + // deferred); fsz%8 ∈ {3,5,6,7} would need shift- + // store and is also unsupported by #4 — falls + // through to the existing AX-only wrongness + // (consistent, tracked as follow-up). + // + // INVARIANT: between cgexpr(N_CALL) and the + // AX/DX/CX stores below, NO instruction may touch + // AX/DX/CX. The BX reload is safe; any other + // emission added here will silently corrupt the + // return value. + let callwhole: bool = false; + if (!nested) { + if (fieldnode.lhs != nil) { + if (fieldnode.lhs.kind == nkind.N_CALL) { + if (fi.tnode != nil) { + if (fi.tnode.kind == nkind.N_TNAME) { + if (primsize(fi.tnode.str) == 0) { + let csi: *structinfo = structlookup(c, fi.tnode.str); + if (csi != nil) { + // Use the inner struct's + // NATURAL size (no 8B slot + // rounding) so MOVL/MOVW/ + // MOVB tail dispatch matches + // cstage's fl->type->size + // (which is natural per + // check.c). fi.fsz here is + // wwstage's slot-padded + // totsize — using it would + // emit 2× MOVQ where cstage + // emits MOVQ+MOVL for a + // 12B inner, etc. (task #15 + // territory; sidestepped + // locally.) + let cfsz: i32 = structnaturalsize(csi); + let crem: i32 = cfsz - (cfsz / 8) * 8; + if (cfsz <= 24) { + if (crem == 0 || crem == 1 + || crem == 2 || crem == 4) { + cgexpr(c, fieldnode.lhs); + if (mode == 1) { + emitline("\tMOVQ\t"); + emitoff(srcoff: i64); + emitline("(BP), BX\n"); + }; + if (mode == 2) { + emitline("\tLEAQ\t"); + emitsymname(c, srcname); + emitline("(SB), BX\n"); + }; + let full: i32 = cfsz / 8; + let ci: i32 = 0; + for (ci < full) { + let r: str = "AX"; + if (ci == 1) { r = "DX"; }; + if (ci == 2) { r = "CX"; }; + emitline("\tMOVQ\t"); + emitline(r); + emitline(", "); + if (mode == 0) { + emitoff((disp + fi.foff + ci * 8): i64); + emitline("(BP)\n"); + } else { + emitdispreg((disp + fi.foff + ci * 8): i64, basereg); + emitline("\n"); + }; + ci += 1; + }; + if (crem > 0) { + let top: str = "MOVB"; + if (crem == 4) { top = "MOVL"; }; + if (crem == 2) { top = "MOVW"; }; + let tr: str = "AX"; + if (full == 1) { tr = "DX"; }; + if (full == 2) { tr = "CX"; }; + emitline("\t"); + emitline(top); + emitline("\t"); + emitline(tr); + emitline(", "); + if (mode == 0) { + emitoff((disp + fi.foff + full * 8): i64); + emitline("(BP)\n"); + } else { + emitdispreg((disp + fi.foff + full * 8): i64, basereg); + emitline("\n"); + }; + }; + callwhole = true; + }; + }; + }; + }; + }; + }; + }; + }; + }; if (nested) { fi = nil; + } else if (callwhole) { + fi = nil; } else { cgexpr(c, fieldnode.lhs); // For non-BP modes, cgexpr just clobbered diff --git a/test/wcc/705_nested_call_rhs.c b/test/wcc/705_nested_call_rhs.c new file mode 100644 index 00000000..3f3cdda3 --- /dev/null +++ b/test/wcc/705_nested_call_rhs.c @@ -0,0 +1,362 @@ +/* + * 705_nested_call_rhs — silent zero of nested struct-typed CALL + * value in the structlit-fill helper (task #20). + * + * Sister bug to #17 / #18. #17 introduced the + * `cg_structlit_fill[_bp]` / `cgstructlitfill[bp]` helper to handle + * nested N_STRUCTLIT field values at BP-rel + dot-lhs sites. #18 + * extended it to the four N_ASSIGN N_DOT-lhs flavors (single-dot + * local/ptr/global, chained-dot through *struct/global). Both fixes + * targeted nested N_STRUCTLIT only. + * + * #20 covers the THIRD silent miscompile: a struct-typed field + * whose VALUE is itself an N_CALL (call returning a struct ≤24B per + * #4's AX/DX/CX cgreturn ABI). Pre-#20 the cgexpr-then-AX-store + * fallthrough inside the helper landed AX = first qword only and + * silently dropped the trailing bytes (DX/CX never made it to the + * destination). + * + * ```ww + * let o: outer = outer { m = mki(), x = 20i64 }; + * // Pre-#20: o.m's first 8 bytes = AX from mki(); rest silently 0. + * ``` + * + * Fix: a new per-field branch in the helper (between the nested- + * STRUCTLIT recursion and the scalar cgexpr-then-AX-store) detects + * `fu->kind == TY_STRUCT && f->lhs->kind == N_CALL` and emits the + * full AX/DX/CX → MOVQ x full + MOVL/MOVW/MOVB tail sequence per + * #4's receive shape. Guard `fsz <= 24 && fsz%8 ∈ {0,1,2,4}` mirrors + * #4 — >24B and fsz%8 ∈ {3,5,6,7} fall through (sret / shift-store + * not yet wired; tracked as a follow-up). + * + * Coverage: + * - tail dispatch: 8/16/24 (tail==0), 12 (MOVL tail==4), + * 10 (MOVW tail==2 — load-bearing), 9 (MOVB tail==1). + * - dst modes: BP-rel (N_LET initializer), PTR_LOCAL (`p.f = ...` + * where p: *holder). Global + chained dst already covered by + * #18's per-mode BX-reload tests; the call branch reuses the + * same reload cadence so a subset suffices. + * - depth: one shallow (call directly under outer literal), one + * 3-deep (call in a literal in a literal under outer) to pin + * `disp + foff` threading through the helper's recursion. + * + * Each row pins: + * - cstage value correctness (process exit code). + * - wwstage value correctness (when ww_ww exists). + * - cstage vs wwstage byte-identical .s output (catches drift). + * The MOVW-tail==2 row is the load-bearing one — that's where + * a divergence between the two stages' new branches is most + * plausible. + */ +#include +#include +#include +#include +#include +#include + +static int +runwait(const char *cmd) +{ + int rc = system(cmd); + if (rc == -1) return -1; + if (WIFEXITED(rc)) return WEXITSTATUS(rc); + return -1; +} + +struct row { const char *label; const char *src; int want; }; + +static const struct row rows[] = { + /* BP-rel, inner struct 8B (one i64 field). fsz=8, full=1, + * tail=0 — AX only, no tail. Pre-#20: o.m.a still landed + * (it WAS in AX). Bug masked at this size. Want: 7+20 = 27. */ + { "bp_call_8b", + "type inner = struct { a: i64 };\n" + "type outer = struct { m: inner, x: i64 };\n" + "fn mki() inner = { return inner { a = 7i64 }; };\n" + "fn main() i32 = {\n" + " let o: outer = outer { m = mki(), x = 20i64 };\n" + " return (o.m.a + o.x): i32;\n" + "};\n", + 27 }, + /* BP-rel, inner struct 16B (two i64 fields). fsz=16, full=2, + * tail=0 — AX + DX, no tail. Pre-#20: o.m.b silently 0. + * Want: 7+8+20 = 35. */ + { "bp_call_16b", + "type inner = struct { a: i64, b: i64 };\n" + "type outer = struct { m: inner, x: i64 };\n" + "fn mki() inner = { return inner { a = 7i64, b = 8i64 }; };\n" + "fn main() i32 = {\n" + " let o: outer = outer { m = mki(), x = 20i64 };\n" + " return (o.m.a + o.m.b + o.x): i32;\n" + "};\n", + 35 }, + /* BP-rel, inner struct 24B (three i64 fields, the headline + * #4 ABI shape). fsz=24, full=3, tail=0 — AX + DX + CX. + * Pre-#20: o.m.b and o.m.c silently 0. Want: 1+2+3+20 = 26. */ + { "bp_call_24b", + "type inner = struct { a: i64, b: i64, c: i64 };\n" + "type outer = struct { m: inner, x: i64 };\n" + "fn mki() inner = { return inner { a = 1i64, b = 2i64, c = 3i64 }; };\n" + "fn main() i32 = {\n" + " let o: outer = outer { m = mki(), x = 20i64 };\n" + " return (o.m.a + o.m.b + o.m.c + o.x): i32;\n" + "};\n", + 26 }, + /* BP-rel, inner 12B (three i32 fields, maxalign=4 → no + * trailing pad). fsz=12, full=1, tail=4 — AX MOVQ + DX MOVL. + * Pins the MOVL-tail dispatch. + * + * The `x: i64` outer field forces outer.maxalign=8 so cstage + * and wwstage agree on outer.totsize (24). Without that, + * wwstage rounds outer.totsize up to 8 (task #15 pre-existing + * struct-sizing divergence) and the BP offsets in the asm + * diverge across stages. Sidestep, not a fix for #15. Want: + * 4+5+6+20 = 35. */ + { "bp_call_12b_movl_tail", + "type inner = struct { a: i32, b: i32, c: i32 };\n" + "type outer = struct { m: inner, x: i64 };\n" + "fn mki() inner = { return inner { a = 4, b = 5, c = 6 }; };\n" + "fn main() i32 = {\n" + " let o: outer = outer { m = mki(), x = 20i64 };\n" + " return (o.m.a: i64 + o.m.b: i64 + o.m.c: i64 + o.x): i32;\n" + "};\n", + 35 }, + /* BP-rel, inner 10B (five i16 fields, maxalign=2 → no + * trailing pad). fsz=10, full=1, tail=2 — AX MOVQ + DX MOVW. + * Rob's load-bearing row: pins MOVW emission on both stages. + * A divergence (cstage emits MOVL/MOVQ tail, wwstage emits + * MOVW) would fail the asm-diff. Same `x: i64` outer-maxalign + * sidestep as the 12B row. Want: 2+3+5+7+11+30 = 58. + * + * Note: the producer (mki's cgreturn) still uses the {1→MOVB, + * 4→MOVL, else MOVQ} field-store dispatch from #17 (task #13). + * For i16 fields this stomps 8 bytes per write, but write- + * order at monotonically increasing field offsets means each + * i16's low 2 bytes stay intact (later MOVQs only clobber + * higher offsets). Receive side reads only the low 2 bytes + * via MOVW tail — so the round-trip value is correct. */ + { "bp_call_10b_movw_tail", + "type inner = struct { a: i16, b: i16, c: i16, d: i16, e: i16 };\n" + "type outer = struct { m: inner, x: i64 };\n" + "fn mki() inner = {\n" + " return inner { a = 2i16, b = 3i16, c = 5i16, d = 7i16, e = 11i16 };\n" + "};\n" + "fn main() i32 = {\n" + " let o: outer = outer { m = mki(), x = 30i64 };\n" + " return (o.m.a: i64 + o.m.b: i64 + o.m.c: i64 + o.m.d: i64\n" + " + o.m.e: i64 + o.x): i32;\n" + "};\n", + 58 }, + /* BP-rel, inner 9B (nine i8 fields, maxalign=1 → no + * trailing pad). fsz=9, full=1, tail=1 — AX MOVQ + DX MOVB. + * Pins the MOVB-tail dispatch. Same `x: i64` outer-maxalign + * sidestep. Want: 1+2+3+4+5+6+7+8+9+40 = 85. */ + { "bp_call_9b_movb_tail", + "type inner = struct {\n" + " a: i8, b: i8, c: i8, d: i8, e: i8,\n" + " f: i8, g: i8, h: i8, i: i8\n" + "};\n" + "type outer = struct { m: inner, x: i64 };\n" + "fn mki() inner = {\n" + " return inner {\n" + " a = 1i8, b = 2i8, c = 3i8, d = 4i8, e = 5i8,\n" + " f = 6i8, g = 7i8, h = 8i8, i = 9i8\n" + " };\n" + "};\n" + "fn main() i32 = {\n" + " let o: outer = outer { m = mki(), x = 40i64 };\n" + " return (o.m.a: i64 + o.m.b: i64 + o.m.c: i64 + o.m.d: i64\n" + " + o.m.e: i64 + o.m.f: i64 + o.m.g: i64 + o.m.h: i64\n" + " + o.m.i: i64 + o.x): i32;\n" + "};\n", + 85 }, + /* PTR_LOCAL: `p.f = outer { m = mki(), x = ... }` where + * p: *holder. Exercises mode=DST_PTR_LOCAL — helper reloads + * BX from srcoff(BP) before the AX/DX/CX stores (cgexpr + * clobbers BX during the call). 16B inner pins the BX reload + * + 2-MOVQ store sequence. Want: 13+17+50 = 80. */ + { "ptrlocal_call_16b", + "type inner = struct { a: i64, b: i64 };\n" + "type outer = struct { m: inner, x: i64 };\n" + "type holder = struct { f: outer };\n" + "fn mki() inner = { return inner { a = 13i64, b = 17i64 }; };\n" + "fn main() i32 = {\n" + " let h: holder;\n" + " let p: *holder = &h;\n" + " p.f = outer { m = mki(), x = 50i64 };\n" + " return (p.f.m.a + p.f.m.b + p.f.x): i32;\n" + "};\n", + 80 }, + /* BP-rel 3-deep: call buried two levels under the outer + * literal. Pins that the `disp + foff` accumulator threads + * correctly through the helper's recursion into the + * call-rhs branch. Inner_in is 16B; middle wraps it + + * an i64; outer wraps middle + an i64. Want: 9+11+30+50 = 100. */ + { "bp_call_3deep_16b", + "type leaf = struct { a: i64, b: i64 };\n" + "type middle = struct { in: leaf, t: i64 };\n" + "type outer = struct { m: middle, x: i64 };\n" + "fn mki() leaf = { return leaf { a = 9i64, b = 11i64 }; };\n" + "fn main() i32 = {\n" + " let o: outer = outer {\n" + " m = middle { in = mki(), t = 30i64 },\n" + " x = 50i64\n" + " };\n" + " return (o.m.in.a + o.m.in.b + o.m.t + o.x): i32;\n" + "};\n", + 100 }, +}; + +/* run_driver — compile r->src via the given driver and exec; return + * the process exit code. Mirror of 703/704. */ +static int +run_driver(const char *driver, const struct row *r, int i) +{ + char src[64], tmpdir[64], cmd[1024]; + snprintf(src, sizeof src, "/tmp/wcnc_%d_%d.ww", getpid(), i); + snprintf(tmpdir, sizeof tmpdir, "/tmp/wcnc_%d_d_%d", getpid(), i); + + FILE *f = fopen(src, "wb"); + if (!f) return -1; + fputs(r->src, f); + fclose(f); + + mkdir(tmpdir, 0755); + snprintf(cmd, sizeof cmd, "cd %s && %s build %s", + tmpdir, driver, src); + if (runwait(cmd) != 0) { + fprintf(stderr, "row[%s]: build via %s failed\n", + r->label, driver); + unlink(src); rmdir(tmpdir); + return -1; + } + + const char *base = strrchr(src, '/'); + base = base ? base + 1 : src; + char outbin[128]; + snprintf(outbin, sizeof outbin, "%s/%s", tmpdir, base); + char *dot = strrchr(outbin, '.'); + if (dot && strcmp(dot, ".ww") == 0) *dot = '\0'; + int got = runwait(outbin); + + unlink(src); unlink(outbin); rmdir(tmpdir); + return got; +} + +/* asm_byte_identical — generate .s via cstage's w6c and wwstage's + * w6c_ww and diff. Mirror of 703/704. */ +static int +asm_byte_identical(const char *bin, const struct row *r, int i) +{ + char src[64], cs[64], ws[64], cmd[1024]; + snprintf(src, sizeof src, "/tmp/wcnc_asm_%d_%d.ww", getpid(), i); + snprintf(cs, sizeof cs, "/tmp/wcnc_asm_%d_%d_c.s", getpid(), i); + snprintf(ws, sizeof ws, "/tmp/wcnc_asm_%d_%d_w.s", getpid(), i); + + FILE *f = fopen(src, "wb"); + if (!f) return -1; + fputs(r->src, f); + fclose(f); + + snprintf(cmd, sizeof cmd, "%s/w6c -o %s %s 2>/dev/null", bin, cs, src); + if (runwait(cmd) != 0) { + fprintf(stderr, "row[%s]: w6c errored\n", r->label); + unlink(src); + return -1; + } + snprintf(cmd, sizeof cmd, "%s/w6c_ww -o %s %s 2>/dev/null", + bin, ws, src); + if (runwait(cmd) != 0) { + fprintf(stderr, "row[%s]: w6c_ww errored\n", r->label); + unlink(src); unlink(cs); + return -1; + } + + FILE *fc = fopen(cs, "rb"); + FILE *fw = fopen(ws, "rb"); + int rc = 0; + if (!fc || !fw) { + rc = -1; + } else { + for (;;) { + int a = fgetc(fc); + int b = fgetc(fw); + if (a != b) { rc = -1; break; } + if (a == EOF) break; + } + } + if (fc) fclose(fc); + if (fw) fclose(fw); + if (rc != 0) + fprintf(stderr, "row[%s]: cstage vs wwstage asm differs\n", + r->label); + unlink(src); unlink(cs); unlink(ws); + return rc; +} + +int +main(void) +{ + const char *bin = getenv("BIN"); + if (!bin) bin = "out/bin"; + char absbin[1024]; + if (bin[0] != '/') { + char cwd[1024]; + if (getcwd(cwd, sizeof cwd) == NULL) return 1; + snprintf(absbin, sizeof absbin, "%s/%s", cwd, bin); + bin = absbin; + } + + char cdrv[1024]; + snprintf(cdrv, sizeof cdrv, "%s/ww", bin); + char wdrv[1024]; + snprintf(wdrv, sizeof wdrv, "%s/ww_ww", bin); + + struct { const char *name; const char *path; int gated_on_existence; } + drivers[] = { + { "cstage", cdrv, 0 }, + { "wwstage", wdrv, 1 }, + { NULL, NULL, 0 }, + }; + + int n = (int)(sizeof rows / sizeof rows[0]); + int total = 0, fail = 0; + + for (int d = 0; drivers[d].name; d++) { + if (drivers[d].gated_on_existence + && access(drivers[d].path, X_OK) != 0) { + fprintf(stderr, "nested_call_rhs: skip %s (no %s)\n", + drivers[d].name, drivers[d].path); + continue; + } + for (int i = 0; i < n; i++) { + int got = run_driver(drivers[d].path, &rows[i], i); + total++; + if (got != rows[i].want) { + fprintf(stderr, + "nested_call_rhs[%s][%s]: exit=%d want=%d\n", + drivers[d].name, rows[i].label, + got, rows[i].want); + fail++; + } + } + } + + if (access(wdrv, X_OK) == 0) { + for (int i = 0; i < n; i++) { + total++; + if (asm_byte_identical(bin, &rows[i], i) != 0) + fail++; + } + } + + if (fail) { + fprintf(stderr, + "nested_call_rhs: %d/%d fixtures failed\n", + fail, total); + return 1; + } + printf("nested_call_rhs: %d/%d ok\n", total, total); + return 0; +}