Files
ww/test/wcc/683_arr_strslice_elem.c
Hojun-Cho ce3a25a0b4 test: contain sepwork scratch per-driver tmpdir, fix /tmp+in-repo leak (#8)
The wcc test drivers ran `ww build <bare-/tmp src>` with no -o, so the
compiler's <stem>.sepwork scratch landed beside the source and was never
cleaned: unbounded /tmp growth (2195 stale dirs observed) that fills tmpfs
and fabricates phantom test failures + silent harness aborts, and for
in-repo fixture builds leaked .sepwork into the tracked tree.

Each leaking build now writes its source + output inside a per-invocation
tmpdir, passes -o <tmpdir>/<stem> so the .sepwork lands inside it, and
rm -rf's the tmpdir on every exit path -- including fopen-fail and the
expected-fail reject builds (scratch is mkdir'd before the build can fail).
`ww run` and explicit-`-o`/byte-id helpers are left as-is; the 990/993
byte-id comparison logic is byte-for-byte unchanged.

Two items filed separately (this commit holds the no-Makefile / no-main.c
rail):
- #13: a stale <src>.s byte-id readback (749) silently no-ops since
  separate-compile emits .s to <ostem>.sepwork/__root.s; documented inline.
- #14: build-system Makefile recipes build selfhost/cmd/*/main.ww with no
  -o and leak main.sepwork in-tree (bounded, gitignored; own commit).

One concern -- sepwork leak hygiene -- across 228 drivers; uniform
transform applied per-file and two-round reviewed. make test: all 402
passed, zero net-new /tmp scratch, zero test-driven in-repo .sepwork.
2026-06-22 23:29:39 +09:00

399 lines
13 KiB
C

/*
* 683_arr_strslice_elem — cstage and wwstage agree, byte-for-byte and
* at runtime, that an `[N][]u8` / `[N]str` array-LITERAL init copies the
* FULL 24B {ptr,len,cap} header of every element (task #20, the #270
* aggregate-element-store family's str/slice arm).
*
* The bug: cgen's N_LET / N_ARRLIT per-element store lowered each str/
* slice element's header into AX=ptr/BX=len/CX=cap (cgexpr) but stored
* only some words — a slice element fell through to the scalar 1-word
* MOVQ (dropping .len AND .cap), and a str element stored 2 words
* (dropping .cap, latent). Each element is 24B and must be copied
* whole. wwstage was worse: a slice element matched no esz branch, so
* esz stayed the 8 sentinel — the per-element stride collapsed (element
* i+1 overwrote element i's tail), the -96-vs-80 cs!=ww frame
* divergence. struct/array/tuple elements already copied correctly via
* the #270-1c is_agg multi-word path; str/slice were the documented
* follow-up (cgen.c:9037-9042, cgenstmt.ww deferral comment).
*
* The fix (BOTH stages, converged byte-identical): cstage adds
* is_slice_el = type_isslice(esub) and stores 3 words (incl CX->base+16)
* for `is_str_el || is_slice_el`; wwstage adds isslicel (TY_SLICE ->
* esz = esubti.size, fixing the stride) and the matching 3-word store.
*
* Cap is validated BOTH ways: a WHOLE-ELEMENT COPY (`let q = t[i]; q.cap`)
* AND a DIRECT `t[i].cap` read. The direct read was a SEPARATE bug (task
* #13, the #20 store's read-sibling): the `.cap` field-extract on an
* INDEXED slice/str element returned .ptr on cstage (it shuffled only
* .len BX→AX, leaving AX=.ptr for cap) and emitted NO read on wwstage
* (the cgdot non-ident catch-all handled only .ptr/.len) — divergent.
* Fix (BOTH stages, byte-id): cgexpr leaves the full {ptr,len,cap} header
* via cgslicehdr for an indexed element, so .cap shuffles CX→AX, the twin
* of the .len BX→AX shuffle. The direct-cap rows must equal the
* whole-element-copy oracle (which #20 made correct).
*
* Mutation-sanity (the per-word coverage): the .len rows fail if the
* store drops the .len word (the pre-fix slice 1-word store), and the
* cap-via-copy rows fail if it drops the .cap word (the pre-fix str
* 2-word store) — so a regression to a 1-word or 2-word store is caught.
*
* row | shape | want
* -----------------+------------------------------------+--------------
* slice_len1 | [2][]u8, return t[1].len. Pre-fix | 2 + byte-id
* | slice 1-word store dropped .len. |
* slice_len0 | [2][]u8, return t[0].len. | 3 + byte-id
* slice_stride3 | [3][]u8, return t[2].len. Exercises| 7 + byte-id
* | the 24B per-element stride (the |
* | wwstage 8-sentinel frame-offset |
* | bug overran into the wrong slot). |
* slice_cap_copy | [2][]u8 w/ caps 7,6; let q=t[0]; | 7 + byte-id
* | return q.cap. Validates the stored |
* | .cap word (pre-fix dropped). |
* slice_ptr | [2][]u8 over a backing array; let | 4 + byte-id
* | q=t[1]; return q[0]. Pins .ptr |
* | stored correctly (deref the elem). |
* str_len1 | [2]str=[a,b], return t[1].len. | 2 + byte-id
* str_cap_copy | [2]str=["abcde","xy"]; let q=t[0]; | 5 + byte-id
* | return q.cap. The latent str cap- |
* | drop (pre-fix garbage); cap=len=5 |
* | for a static literal. |
* struct_elem | [2]Pt struct literal, return | 4 + byte-id
* | t[1].y. Regression pin: the is_agg |
* | multi-word path is untouched by |
* | the str/slice branch. |
*
* Exit-code rows confirm both stages run correctly. The asm-byte-id rows
* pin the symmetric 3-word store (cstage == wwstage); pre-fix wwstage
* mis-strided (slice esz=8) and under-copied, so the diff was non-empty.
*/
#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <sys/stat.h>
#include <sys/wait.h>
static int
runwait(const char *cmd)
{
int rc = system(cmd);
if (rc == -1) return -1;
if (WIFEXITED(rc)) return WEXITSTATUS(rc);
return -1;
}
/* want == BYTEID_ONLY: the row's runtime value is non-deterministic (a
* link-time address), so only the cstage/wwstage asm-byte-id is asserted;
* the build must still succeed on both stages. Used for the `"abc".cap`
* symmetry edge (a string-literal .cap is meaningless garbage on both
* stages, but rule 10 still requires byte-identical asm). */
#define BYTEID_ONLY (-2147483647 - 1)
struct row { const char *label; const char *src; int want; };
static const struct row rows[] = {
{ "slice_len1",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 8;\n"
"\tlet b: []u8; b.ptr = &hb[0]; b.len = 2; b.cap = 8;\n"
"\tlet t: [2][]u8 = [a, b];\n"
"\treturn t[1].len: i32;\n"
"};\n",
2 },
{ "slice_len0",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 8;\n"
"\tlet b: []u8; b.ptr = &hb[0]; b.len = 2; b.cap = 8;\n"
"\tlet t: [2][]u8 = [a, b];\n"
"\treturn t[0].len: i32;\n"
"};\n",
3 },
{ "slice_stride3",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 8;\n"
"\tlet b: []u8; b.ptr = &hb[0]; b.len = 5; b.cap = 8;\n"
"\tlet c: []u8; c.ptr = &hb[0]; c.len = 7; c.cap = 8;\n"
"\tlet t: [3][]u8 = [a, b, c];\n"
"\treturn t[2].len: i32;\n"
"};\n",
7 },
{ "slice_cap_copy",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 7;\n"
"\tlet b: []u8; b.ptr = &hb[0]; b.len = 2; b.cap = 6;\n"
"\tlet t: [2][]u8 = [a, b];\n"
"\tlet q: []u8 = t[0];\n"
"\treturn q.cap: i32;\n"
"};\n",
7 },
/* #13 direct-read rows: `t[i].cap` straight (no copy via a let).
* Must equal the slice_cap_copy / str_cap_copy oracle. Pre-fix:
* cstage returned .ptr (a heap/stack address, != 7/6), wwstage
* emitted no read (stale AX) — and the two diverged. */
{ "slice_cap_direct0",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 7;\n"
"\tlet b: []u8; b.ptr = &hb[0]; b.len = 2; b.cap = 6;\n"
"\tlet t: [2][]u8 = [a, b];\n"
"\treturn t[0].cap: i32;\n"
"};\n",
7 },
{ "slice_cap_direct1",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 7;\n"
"\tlet b: []u8; b.ptr = &hb[0]; b.len = 2; b.cap = 6;\n"
"\tlet t: [2][]u8 = [a, b];\n"
"\treturn t[1].cap: i32;\n"
"};\n",
6 },
{ "str_cap_direct0",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet a: str = \"abcde\";\n"
"\tlet b: str = \"xy\";\n"
"\tlet t: [2]str = [a, b];\n"
"\treturn t[0].cap: i32;\n"
"};\n",
5 },
{ "str_cap_direct1",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet a: str = \"abcde\";\n"
"\tlet b: str = \"xy\";\n"
"\tlet t: [2]str = [a, b];\n"
"\treturn t[1].cap: i32;\n"
"};\n",
2 },
/* #13 symmetry edge: `.cap` of a BARE string literal. The wwstage
* checker types N_STRLIT as `str` (cstage types it untyped_str), so
* the .cap kind-gate would wrongly shuffle CX→AX on wwstage only —
* but a literal's cgexpr never loads a CX cap, and cstage never
* shuffles it, so both must return AX (ptr) unshuffled. Byte-id only:
* the value is a link-time address (non-deterministic). Pre-fix this
* row's asm DIFFERED (wwstage had the stray MOVQ CX, AX). */
{ "str_lit_cap_symmetry",
"package main;\n"
"export fn main() i32 = {\n"
"\treturn \"abc\".cap: i32;\n"
"};\n",
BYTEID_ONLY },
{ "slice_ptr",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet hb: [8]u8; hb[0] = 9u8; hb[1] = 4u8;\n"
"\tlet a: []u8; a.ptr = &hb[0]; a.len = 3; a.cap = 8;\n"
"\tlet b: []u8; b.ptr = &hb[1]; b.len = 2; b.cap = 8;\n"
"\tlet t: [2][]u8 = [a, b];\n"
"\tlet q: []u8 = t[1];\n"
"\treturn q[0]: i32;\n"
"};\n",
4 },
{ "str_len1",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet a: str = \"abc\";\n"
"\tlet b: str = \"de\";\n"
"\tlet t: [2]str = [a, b];\n"
"\treturn t[1].len: i32;\n"
"};\n",
2 },
{ "str_cap_copy",
"package main;\n"
"export fn main() i32 = {\n"
"\tlet a: str = \"abcde\";\n"
"\tlet b: str = \"xy\";\n"
"\tlet t: [2]str = [a, b];\n"
"\tlet q: str = t[0];\n"
"\treturn q.cap: i32;\n"
"};\n",
5 },
/* Regression pin: a [N]struct element copies multi-word via the
* pre-existing #270-1c is_agg path, NOT the new str/slice 3-word
* header branch. Proves the str/slice esz/store change leaves the
* is_agg element path untouched (byte-id holds for it too). */
{ "struct_elem",
"package main;\n"
"type Pt = struct { x: i32, y: i32 };\n"
"export fn main() i32 = {\n"
"\tlet t: [2]Pt = [Pt { x = 1, y = 2 }, Pt { x = 3, y = 4 }];\n"
"\treturn t[1].y: i32;\n"
"};\n",
4 },
};
static int
run_driver(const char *driver, const struct row *r, int i)
{
char tmpdir[64], src[128], outbin[128], rmcmd[160], cmd[1024];
snprintf(tmpdir, sizeof tmpdir, "/tmp/asse_%d_d_%d", getpid(), i);
mkdir(tmpdir, 0755);
snprintf(src, sizeof src, "%s/asse_%d_%d.ww", tmpdir, getpid(), i);
snprintf(outbin, sizeof outbin, "%s/asse_%d_%d", tmpdir, getpid(), i);
snprintf(rmcmd, sizeof rmcmd, "rm -rf %s", tmpdir);
FILE *f = fopen(src, "wb");
if (!f) { runwait(rmcmd); return -1; }
fputs(r->src, f);
fclose(f);
snprintf(cmd, sizeof cmd, "%s build -o %s %s 2>/dev/null",
driver, outbin, src);
if (runwait(cmd) != 0) {
fprintf(stderr, "row[%s]: build via %s failed\n",
r->label, driver);
runwait(rmcmd);
return -1;
}
int got = runwait(outbin);
runwait(rmcmd);
return got;
}
/* asm_byte_identical — generate .s via cstage's w6c and wwstage's
* w6c_ww and diff. The regression-pinning row for #20: pre-fix wwstage
* mis-strided the slice element (esz=8) and under-copied, so the diff
* was non-empty; the converged 3-word store makes them identical. */
static int
asm_byte_identical(const char *bin, const struct row *r, int i)
{
char src[64], cs[64], ws[64], cmd[1024];
snprintf(src, sizeof src, "/tmp/asse_asm_%d_%d.ww", getpid(), i);
snprintf(cs, sizeof cs, "/tmp/asse_asm_%d_%d_c.s", getpid(), i);
snprintf(ws, sizeof ws, "/tmp/asse_asm_%d_%d_w.s", getpid(), i);
FILE *f = fopen(src, "wb");
if (!f) return -1;
fputs(r->src, f);
fclose(f);
snprintf(cmd, sizeof cmd, "%s/w6c -o %s %s 2>/dev/null", bin, cs, src);
if (runwait(cmd) != 0) {
fprintf(stderr, "row[%s]: w6c errored\n", r->label);
unlink(src);
return -1;
}
snprintf(cmd, sizeof cmd, "%s/w6c_ww -o %s %s 2>/dev/null",
bin, ws, src);
if (runwait(cmd) != 0) {
fprintf(stderr, "row[%s]: w6c_ww errored\n", r->label);
unlink(src); unlink(cs);
return -1;
}
FILE *fc = fopen(cs, "rb");
FILE *fw = fopen(ws, "rb");
int rc = 0;
if (!fc || !fw) {
rc = -1;
} else {
for (;;) {
int a = fgetc(fc);
int b = fgetc(fw);
if (a != b) { rc = -1; break; }
if (a == EOF) break;
}
}
if (fc) fclose(fc);
if (fw) fclose(fw);
if (rc != 0)
fprintf(stderr, "row[%s]: cstage vs wwstage asm differs\n",
r->label);
unlink(src); unlink(cs); unlink(ws);
return rc;
}
int
main(void)
{
const char *bin = getenv("BIN");
if (!bin) bin = "out/bin";
char absbin[1024];
if (bin[0] != '/') {
char cwd[1024];
if (getcwd(cwd, sizeof cwd) == NULL) return 1;
snprintf(absbin, sizeof absbin, "%s/%s", cwd, bin);
bin = absbin;
}
char cdrv[1024];
snprintf(cdrv, sizeof cdrv, "%s/ww", bin);
char wdrv[1024];
snprintf(wdrv, sizeof wdrv, "%s/ww_ww", bin);
struct { const char *name; const char *path; int gated_on_existence; }
drivers[] = {
{ "cstage", cdrv, 0 },
{ "wwstage", wdrv, 1 },
{ NULL, NULL, 0 },
};
int n = (int)(sizeof rows / sizeof rows[0]);
int total = 0, fail = 0;
for (int d = 0; drivers[d].name; d++) {
if (drivers[d].gated_on_existence
&& access(drivers[d].path, X_OK) != 0) {
fprintf(stderr, "arr_strslice_elem: skip %s (no %s)\n",
drivers[d].name, drivers[d].path);
continue;
}
for (int i = 0; i < n; i++) {
int got = run_driver(drivers[d].path, &rows[i], i);
total++;
/* BYTEID_ONLY rows assert asm byte-id below; here only
* the build must succeed (run_driver returns -1 on a
* build/run failure). The exit value is ignored. */
int bad = rows[i].want == BYTEID_ONLY
? (got == -1) : (got != rows[i].want);
if (bad) {
fprintf(stderr,
"arr_strslice_elem[%s][%s]: exit=%d want=%d\n",
drivers[d].name, rows[i].label,
got, rows[i].want);
fail++;
}
}
}
if (access(wdrv, X_OK) == 0) {
for (int i = 0; i < n; i++) {
total++;
if (asm_byte_identical(bin, &rows[i], i) != 0)
fail++;
}
}
if (fail) {
fprintf(stderr,
"arr_strslice_elem: %d/%d fixtures failed\n", fail, total);
return 1;
}
printf("arr_strslice_elem: %d/%d ok\n", total, total);
return 0;
}