wcc: float array-element loads to X0 + indexed-float consumer (#119)

cgindex's element-load sites ended in the integer loadopsz (MOVQ/MOVL
into AX), with no float branch — so an f32/f64 array element landed in
a GPR while the consumer's ADDSD/MOVSD read a stale X0. Add a float-
element branch (MOVSS f32 / MOVSD f64 into X0) at all three wwstage
cgindex sites (global, baselocal, fallback) and both cstage N_INDEX
element-load sites, deriving float-ness from the SAME stamped element
tinfo the esz already reads: new elemisfloatc/elemisf32c helpers
(mirroring elemissignedc) for ident bases, typeisfloat/typeisf32(n.type_)
for N_DOT/N_INDEX bases — never a fresh node-stamp that could hit an
unstamped base (#121).

The load fix cannot land alone: the wwstage consumer (cgbin/cgcast)
classified an indexed float operand as INTEGER (no exprfloatkind N_INDEX
arm) and fell to PUSHQ/ADDQ/MOVSXD, while the cstage read the stamped
operand type and used ADDSD/CVTTSD2SI. That divergence is pre-existing
on master (proven: master cs vs ww already differ on `a[0]+a[1]`),
contradicting the original "consumer already expects X0, cs==ww"
premise; load-only would leave the wwstage incoherent (value in X0,
consumed from AX) and still cs!=ww. So this also adds the exprfloatkind
N_INDEX arm — safe because the index-result type_ IS checker-stamped
(cgindex reads it for esz), unlike the unstamped-N_MLET case deferred
under #121. With both, f64 arrays are runtime-correct and both stages
emit byte-identical asm.

946_floatarr_run: f64 element add / trunc / non-adjacent index assert
the value + cs==ww; the f32 row asserts cs==ww only — its runtime value
is blocked by a SEPARATE store-side bug (f32 array-element store writes
AX raw double low-bits instead of CVTSD2SS-narrowed X0), filed as
#119-store. Regen w6c/wwdump combined.ww (cgenexpr.ww + cgenutil.ww
embedded).
This commit is contained in:
2026-05-26 13:06:42 +09:00
parent a1dff13ec1
commit 0917fee48d
7 changed files with 499 additions and 3 deletions

View File

@@ -11658,6 +11658,34 @@ fn elemissignedc(c: *cgen, t: *node) bool = {
return typeissigned(ti.sub);
};
// elemisfloatc — given an indexable type-AST (`*T`, `[]T`, `[N]T`), is
// its element an f32/f64? Used by cgindex to route the element load to
// MOVSS/MOVSD into X0 instead of the integer loadopsz into AX (#119 —
// the array-element twin of the scalar-float global load at cgen.c:
// 2014). Reads through the stamped tinfo, peeling TY_NAMED before the
// .sub read exactly as elemissignedc does (#64/#65). Float-ness comes
// from the SAME tinfo the esz already reads — never a fresh node-stamp
// (the unstamped-base trap that broke the exprfloatkind collapse, #121).
fn elemisfloatc(c: *cgen, t: *node) bool = {
if (t == nil) { return false; };
let ti: *tinfo = t.type_: *tinfo;
if (ti == nil) { return false; };
for (ti != nil && ti.kind == tykind.TY_NAMED) { ti = ti.under; };
if (ti == nil) { return false; };
return typeisfloat(ti.sub);
};
// elemisf32c — narrower elemisfloatc: true only when the element is f32,
// so cgindex picks MOVSS over MOVSD at the #119 element load.
fn elemisf32c(c: *cgen, t: *node) bool = {
if (t == nil) { return false; };
let ti: *tinfo = t.type_: *tinfo;
if (ti == nil) { return false; };
for (ti != nil && ti.kind == tykind.TY_NAMED) { ti = ti.under; };
if (ti == nil) { return false; };
return typeisf32(ti.sub);
};
// fieldissignedc — does this field/element type-AST need sign-
// extension on a sub-word load? One-liner via typeissigned (cstage
// cgen.c:240 `fld_issigned` SSoT). t.type_ is stamped at check.ww
@@ -12732,6 +12760,21 @@ export fn exprfloatkind(c: *cgen, n: *node) i32 = {
if (isfloattype(c, n.rhs)) { return 2; };
return 0;
};
if (k == nkind.N_INDEX) {
// #119: a float array/slice element feeds cgbin / cgcast through
// X0 (the #119 load is MOVSS/MOVSD into X0). Without this arm the
// wwstage consumer falls to integer (PUSHQ/ADDQ, MOVSXD) while
// the cstage reads the stamped operand type and uses ADDSD/
// CVTTSD2SI — a rule-10 divergence the #119 load fix exposes (the
// consumer was never aligned for indexed float operands). The
// index-result type_ IS checker-stamped (cgindex reads it for
// esz at the N_DOT/N_INDEX-base arms), so this is NOT the
// unstamped-N_MLET-base trap that deferred the broader collapse
// (#121) — only the always-stamped N_INDEX case is classified.
if (isf32type(c, n)) { return 1; };
if (isfloattype(c, n)) { return 2; };
return 0;
};
if (k == nkind.N_IDENT) {
let lc: *local = localfindnode(c, n.str);
if (lc != nil) {
@@ -14944,6 +14987,11 @@ fn cgindex(c: *cgen, n: *node) void = {
let idx: *node = n.rhs;
let esz: i32 = 8;
let signed_elem: bool = false;
// #119: float element loads route to MOVSS/MOVSD into X0, not the
// integer loadopsz into AX. float_elem/f32_elem are set per-branch
// from the SAME tinfo esz reads — never a fresh node-stamp (#121).
let float_elem: bool = false;
let f32_elem: bool = false;
// #1/Phase 3: str and slice are both 24B (and a >16B struct is
// 24B+ too), so the header branches below MUST gate on KIND
// (elemisstr/elemisslice, mirroring cstage's elem_is_str||
@@ -14969,6 +15017,8 @@ fn cgindex(c: *cgen, n: *node) void = {
if (baselocal != nil) {
esz = elemsizeofc(c, baselocal.tnode);
signed_elem = elemissignedc(c, baselocal.tnode);
float_elem = elemisfloatc(c, baselocal.tnode);
f32_elem = elemisf32c(c, baselocal.tnode);
} else {
let tn: *node = letvartnode(c, bn);
if (tn != nil) {
@@ -14977,12 +15027,16 @@ fn cgindex(c: *cgen, n: *node) void = {
globalname = bn;
esz = elemsizeofc(c, tn);
signed_elem = elemissignedc(c, tn);
float_elem = elemisfloatc(c, tn);
f32_elem = elemisf32c(c, tn);
};
if (tn.kind == nkind.N_TPTR) {
isglobalptr = true;
globalname = bn;
esz = elemsizeofc(c, tn);
signed_elem = elemissignedc(c, tn);
float_elem = elemisfloatc(c, tn);
f32_elem = elemisf32c(c, tn);
};
};
};
@@ -14994,7 +15048,7 @@ fn cgindex(c: *cgen, n: *node) void = {
// cgen.c:3517-18). esz-only — N_DOT-base signedness
// stays unset, as before.
let dt: *tinfo = n.type_: *tinfo;
if (dt != nil) { esz = dt.size: i32; elemisstr = typeisstr(dt); elemisslice = typeisslice(dt); };
if (dt != nil) { esz = dt.size: i32; elemisstr = typeisstr(dt); elemisslice = typeisslice(dt); float_elem = typeisfloat(dt); f32_elem = typeisf32(dt); };
} else { if (base.kind == nkind.N_INDEX) {
// #60: chained `names[i][k]` — n.type_ is the checker-
// stamped outer element tinfo (indexresult over the inner
@@ -15005,6 +15059,8 @@ fn cgindex(c: *cgen, n: *node) void = {
if (et != nil) {
esz = et.size: i32;
signed_elem = typeissigned(et);
float_elem = typeisfloat(et);
f32_elem = typeisf32(et);
};
};};};
};
@@ -15086,6 +15142,18 @@ fn cgindex(c: *cgen, n: *node) void = {
cgslicehdr(c, "BX");
return;
};
// #119: float element → MOVSS/MOVSD into X0 (the consumer's
// ADDSD/MOVSD spill machinery already expects X0); the integer
// loadopsz below would leave it in AX and the SSE side reads
// stale. Twin of cgen.c:2014's scalar-float global load.
if (float_elem) {
let fop1: str = "MOVSD";
if (f32_elem) { fop1 = "MOVSS"; };
emitline("\t");
emitline(fop1);
emitline("\t(BX), X0\n");
return;
};
let lop1: str = loadopsz(signed_elem, esz);
emitline("\t");
emitline(lop1);
@@ -15125,6 +15193,15 @@ fn cgindex(c: *cgen, n: *node) void = {
cgslicehdr(c, "BX");
return;
};
// #119: float element → X0 (see the global arm above).
if (float_elem) {
let fop2: str = "MOVSD";
if (f32_elem) { fop2 = "MOVSS"; };
emitline("\t");
emitline(fop2);
emitline("\t(BX), X0\n");
return;
};
let lop2: str = loadopsz(signed_elem, esz);
emitline("\t");
emitline(lop2);
@@ -15156,6 +15233,16 @@ fn cgindex(c: *cgen, n: *node) void = {
cgslicehdr(c, "AX");
return;
};
// #119: float element → X0 (see the global arm above). The base
// address is in AX; MOVSS/MOVSD reads the element into X0.
if (float_elem) {
let fop3: str = "MOVSD";
if (f32_elem) { fop3 = "MOVSS"; };
emitline("\t");
emitline(fop3);
emitline("\t(AX), X0\n");
return;
};
let lop3: str = loadopsz(signed_elem, esz);
emitline("\t");
emitline(lop3);