test: migrate 110_uniesc rows to in-language @test (test/lang pilot)

Proof-of-path for the test-arch reframe: pure-behavior rows become
in-language @test row-tables run by `ww test`, reserving C drivers for
the two irreducible jobs (byte-id + bootstrap). 110_uniesc's 6 value
rows become [N]struct row-tables in a new test/lang/ tree; the trailing
.wwi round-trip stays in C as external-observer residue. Additive — the
C rows are kept; de-dup and make-rewiring land at fold 6 (task #12).
Green on both stages.
This commit is contained in:
2026-06-22 01:48:38 +09:00
parent 70ef7681a6
commit 2f788305b4

76
test/lang/uniesc_test.ww Normal file
View File

@@ -0,0 +1,76 @@
// uniesc_test — in-language behavior rows for \u / \U / \x escapes,
// migrated from test/wcc/110_uniesc_run.c (task #12 proof-of-path). The
// compiler-under-test compiles THIS file, so a miscompile of an escape
// decode fails the assert in-language rather than via an external runner.
//
// Each row's literal is the exact escape form from the C source (not the
// raw multibyte character): the feature under test is the escape-decode
// path (ref/hare/hare/lex/lex.ha:347 lex_unicode + appendrune in
// lex_string), distinct from source-UTF-8 decoding. A rune literal carries
// the raw codepoint; a string literal UTF-8-encodes it into 1-4 bytes; \x
// >=0x80 shares the codepoint path so it widens to its multibyte form.
//
// Carve-out: the C file's trailing .wwi round-trip case (an exported
// wide-rune `def` re-serialized byte-exact) is an external-observer
// residue and stays in C — not migrated here.
package uniesc_test;
import strings;
// One rune row: the escape literal must decode to codepoint `cp`.
type runecase = struct {
lit: rune,
cp: u32,
};
// rune_u_2byte (é -> U+00E9) and rune_U_emoji (\U0001F600 -> U+1F600,
// the >0xFFFF case that forced the int-cast spelling pre-#50).
@test fn rune_escapes() void = {
let rows: [2]runecase = [
runecase { lit = '\u00e9', cp = 233u32 },
runecase { lit = '\U0001F600', cp = 128512u32 },
];
let i: i32 = 0;
for (i < len(rows)) {
assert((rows[i].lit: u32) == rows[i].cp);
i += 1;
};
};
// One string row: the escape literal must UTF-8-encode to `n` bytes
// b0..b3 (unused tail bytes are 0 and guarded by the `n` check).
type strcase = struct {
lit: str,
n: i32,
b0: u8,
b1: u8,
b2: u8,
b3: u8,
};
// str_u_2byte (é -> C3 A9), str_u_3byte (€ -> E2 82 AC),
// str_U_4byte (\U0001F600 -> F0 9F 98 80), str_x_widens (\xe9 -> C3 A9,
// the >=0x80 \x byte that now widens to its 2-byte UTF-8 form).
@test fn string_escapes() void = {
let rows: [4]strcase = [
strcase { lit = "\u00e9", n = 2,
b0 = 0xc3u8, b1 = 0xa9u8, b2 = 0u8, b3 = 0u8 },
strcase { lit = "\u20ac", n = 3,
b0 = 0xe2u8, b1 = 0x82u8, b2 = 0xacu8, b3 = 0u8 },
strcase { lit = "\U0001F600", n = 4,
b0 = 0xf0u8, b1 = 0x9fu8, b2 = 0x98u8, b3 = 0x80u8 },
strcase { lit = "\xe9", n = 2,
b0 = 0xc3u8, b1 = 0xa9u8, b2 = 0u8, b3 = 0u8 },
];
let i: i32 = 0;
for (i < len(rows)) {
let b: []u8 = strings.toutf8(rows[i].lit);
assert(len(b) == rows[i].n);
assert(b[0] == rows[i].b0);
if (rows[i].n > 1) { assert(b[1] == rows[i].b1); };
if (rows[i].n > 2) { assert(b[2] == rows[i].b2); };
if (rows[i].n > 3) { assert(b[3] == rows[i].b3); };
i += 1;
};
};