Replace the cmd/ww + selfhost driver's file-walk import resolver with true directory enumeration. `import encoding.utf8;` now finds the lib/encoding/utf8/ directory and concatenates every *.ww file in it (excluding *test.ww and the driver's *.combined.ww artifacts) in byte-wise sorted order, instead of just finding the single lib/encoding/utf8/utf8.ww file. Mirrors Hare's hare/module/srcs.ha:183 _findsrcs minus tag handling. Lookup order in both stages: (1) <dir>/<dot-as-slash>/ as directory → enumerate. (2) <dir>/<dot-as-slash>.ww as file. The legacy <dir>/<name>/<name>.ww shape from #18's retained divergence is dropped per rule-9 Hare-fidelity — Hare has no foo/foo.ha fallback; a module IS the directory. Symmetric across cstage (cmd/ww/main.c via opendir+qsort+stat) and wwstage (selfhost/cmd/ww/main.ww via existing lib/os.getdents64 + os.stat — no new lib/os surface needed; the rundirtests() walker in main.ww from #18 was the model). Bootstrap ww2.s==ww3.s==ww4.s byte-identical post-change. Bundling justification (rule 11): strict-same-package validation is bundled because the failure mode is dir-enum's own (a non-dir-enum compilation unit cannot trigger mismatch across enumerated files). The natural enforcement site is the driver — the parser can't distinguish dir-enum concat from file-walk concat. Both stages peek each file's first `package <name>;` line in expand_dir / expanddir and exit(1) on mismatch with a precise error pointing at the offending file. Hare's hare/module/srcs.ha:131 has the same constraint via its README gate. Other half of #23 (strict missing-package error tightening — 63 inline-source test wrappers blocker) stays deferred per its filing. Parser side (cmd/wcc/parse.c parseuse + lib/ww/parse/decl.ww parseuse): n->str now carries only the LEAF identifier from a dotted import. With the driver translating the full dotted path to a directory walk, the checker only needs the package bareword (last component) for the N_USE → decl disambiguation walk in check.c's src_imports / decl_mod. Mirrors Hare's `use encoding::utf8;` → `utf8::name` semantics (ref/hare/hare/ast/import.ha:7). Migration: lib/ww/sym.ww drops `import typ; import ast;`; lib/ww/parse/parse.ww drops `import expr; import stmt; import decl;`; lib/ww/lex/lex.ww drops `import tok;` — all sibling imports auto-resolve via the new dir-enum when callers import the package directory. lib/strings/, lib/encoding/utf8/utf8test.ww migrate `import utf8;` → `import encoding.utf8;`. Makefile drops -I lib/encoding/utf8 stopgap from wwdump_ww + w6c_ww. Seven test wrappers (700_e2e, 966_strings_run, 970_fmt_run, 971_log_run, 972_fnmatch_run, 982_getopt_run, 990_selfhost) and 995_self_rebuild drop the -I lib/encoding/utf8 runtime stopgap. Tests: new 737_direnum C wrapper + test/wcc/data/direnum/ fixtures pin (a) cross-pkg multi-file dir-enum build at runtime (both stages must succeed) and (b) strict-same-package mismatch error (both stages must surface "differs from" + exit non-zero). 738_module_decl gains row 6 pinning the n_use->str leaf-only storage post-parser change. Retained workaround at selfhost/cmd/ww/main.ww expanddir loop: `names[i][k]` nested-deref-then-index split into `let nm: *u8 = names[i]; nm[k]` because wwstage cgen miscompiles the chained form (treats inner u8 element as 8B sizeof *u8 instead of 1B sizeof u8: extra MOVQ $8 + IMULQ on the inner index, MOVQ instead of MOVZBQ load). Inline rule-8 WHY comment cites task #24 (wwstage cgen chained-index inner element size on **T). Two-step form routes through the bare-pointer index path which both stages handle byte-identically. Class A wwstage cgen UNDER (chained-index inner element size on **T) surfaced first time the codebase exercises the **T[i][k] shape via enumeratedir() — corpus-coverage-blind landmine pattern, same family as the trio (#27/#28/#31) from STATUS-5. 112/112 ok. ww2 == ww3 == ww4 byte-id holds.
344 lines
11 KiB
Plaintext
344 lines
11 KiB
Plaintext
// utf8test — exercises lib/encoding/utf8. Run with
|
|
// `out/bin/ww run lib/encoding/utf8/utf8test.ww`. Same signalled-
|
|
// then-fail()-with-+10 pattern as hex / base32 / time tests:
|
|
// non-zero exit pinpoints the failing scenario.
|
|
|
|
package utf8;
|
|
|
|
import encoding.utf8;
|
|
import os;
|
|
|
|
let signalled: i32 = 0;
|
|
fn fail() void = { os.exit(signalled + 10); };
|
|
|
|
// ---- runesz: byte length per range ------------------------------------
|
|
// ref/hare/encoding/utf8/rune.ha:5. Boundaries: 0x7F → 1, 0x80 → 2,
|
|
// 0x7FF → 2, 0x800 → 3, 0xFFFF → 3, 0x10000 → 4, 0x10FFFF → 4.
|
|
|
|
@test fn runesz_ranges() void = {
|
|
if (utf8.runesz(0u32: rune) != 1) { fail(); };
|
|
if (utf8.runesz(0x7Fu32: rune) != 1) { fail(); };
|
|
if (utf8.runesz(0x80u32: rune) != 2) { fail(); };
|
|
if (utf8.runesz(0x7FFu32: rune) != 2) { fail(); };
|
|
if (utf8.runesz(0x800u32: rune) != 3) { fail(); };
|
|
if (utf8.runesz(0xFFFFu32: rune) != 3) { fail(); };
|
|
if (utf8.runesz(0x10000u32: rune) != 4) { fail(); };
|
|
if (utf8.runesz(0x10FFFFu32: rune) != 4) { fail(); };
|
|
};
|
|
|
|
// ---- utf8sz: start-byte classification --------------------------------
|
|
// ref/hare/encoding/utf8/rune.ha:15. ASCII → 1; legal multibyte
|
|
// leads → 2/3/4; continuation and >0xF7 → invalid.
|
|
|
|
@test fn utf8sz_classify() void = {
|
|
match (utf8.utf8sz(0u8)) {
|
|
case let n: i32 => { if (n != 1) { fail(); }; };
|
|
case let e: utf8.invalid => { fail(); };
|
|
};
|
|
match (utf8.utf8sz(0x7Fu8)) {
|
|
case let n: i32 => { if (n != 1) { fail(); }; };
|
|
case let e: utf8.invalid => { fail(); };
|
|
};
|
|
match (utf8.utf8sz(0x80u8)) { // continuation
|
|
case let n: i32 => { fail(); };
|
|
case let e: utf8.invalid => void;
|
|
};
|
|
match (utf8.utf8sz(0xC1u8)) { // overlong 2-byte lead
|
|
case let n: i32 => { fail(); };
|
|
case let e: utf8.invalid => void;
|
|
};
|
|
match (utf8.utf8sz(0xC2u8)) {
|
|
case let n: i32 => { if (n != 2) { fail(); }; };
|
|
case let e: utf8.invalid => { fail(); };
|
|
};
|
|
match (utf8.utf8sz(0xE0u8)) {
|
|
case let n: i32 => { if (n != 3) { fail(); }; };
|
|
case let e: utf8.invalid => { fail(); };
|
|
};
|
|
match (utf8.utf8sz(0xF0u8)) {
|
|
case let n: i32 => { if (n != 4) { fail(); }; };
|
|
case let e: utf8.invalid => { fail(); };
|
|
};
|
|
match (utf8.utf8sz(0xF8u8)) { // 5-byte lead — illegal in modern UTF-8
|
|
case let n: i32 => { fail(); };
|
|
case let e: utf8.invalid => void;
|
|
};
|
|
match (utf8.utf8sz(0xFFu8)) {
|
|
case let n: i32 => { fail(); };
|
|
case let e: utf8.invalid => void;
|
|
};
|
|
};
|
|
|
|
// ---- encoderune: all four widths --------------------------------------
|
|
// ref/hare/encoding/utf8/encode.ha:7. Byte vectors come from the
|
|
// Unicode specification (UAX standard examples).
|
|
|
|
@test fn encode_ascii() void = {
|
|
let out: [4]u8;
|
|
let n: i32 = utf8.encoderune(out[0:4], 0x41u32: rune); // 'A'
|
|
if (n != 1) { fail(); };
|
|
if (out[0] != 0x41u8) { fail(); };
|
|
};
|
|
|
|
@test fn encode_two_byte() void = {
|
|
let out: [4]u8;
|
|
let n: i32 = utf8.encoderune(out[0:4], 0xE9u32: rune); // 'é' U+00E9
|
|
if (n != 2) { fail(); };
|
|
if (out[0] != 0xC3u8) { fail(); };
|
|
if (out[1] != 0xA9u8) { fail(); };
|
|
};
|
|
|
|
@test fn encode_three_byte() void = {
|
|
let out: [4]u8;
|
|
let n: i32 = utf8.encoderune(out[0:4], 0x20ACu32: rune); // '€' U+20AC
|
|
if (n != 3) { fail(); };
|
|
if (out[0] != 0xE2u8) { fail(); };
|
|
if (out[1] != 0x82u8) { fail(); };
|
|
if (out[2] != 0xACu8) { fail(); };
|
|
};
|
|
|
|
@test fn encode_four_byte() void = {
|
|
let out: [4]u8;
|
|
let n: i32 = utf8.encoderune(out[0:4], 0x1F980u32: rune); // '🦀' U+1F980
|
|
if (n != 4) { fail(); };
|
|
if (out[0] != 0xF0u8) { fail(); };
|
|
if (out[1] != 0x9Fu8) { fail(); };
|
|
if (out[2] != 0xA6u8) { fail(); };
|
|
if (out[3] != 0x80u8) { fail(); };
|
|
};
|
|
|
|
// ---- decoder.next: valid 1/2/3/4-byte ---------------------------------
|
|
// ref/hare/encoding/utf8/decode.ha:27. Round-trip via the same byte
|
|
// vectors used in encode.
|
|
|
|
@test fn decode_one_byte() void = {
|
|
let src: [1]u8;
|
|
src[0] = 0x41u8;
|
|
let d: utf8.decoder = utf8.decode(src[0:1]);
|
|
match (utf8.next(&d)) {
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
case let r: utf8.invalid => { fail(); };
|
|
case let r: rune => { if (r != 0x41u32: rune) { fail(); }; };
|
|
};
|
|
};
|
|
|
|
@test fn decode_two_byte() void = {
|
|
let src: [2]u8;
|
|
src[0] = 0xC3u8; src[1] = 0xA9u8;
|
|
let d: utf8.decoder = utf8.decode(src[0:2]);
|
|
match (utf8.next(&d)) {
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
case let r: utf8.invalid => { fail(); };
|
|
case let r: rune => { if (r != 0xE9u32: rune) { fail(); }; };
|
|
};
|
|
};
|
|
|
|
@test fn decode_three_byte() void = {
|
|
let src: [3]u8;
|
|
src[0] = 0xE2u8; src[1] = 0x82u8; src[2] = 0xACu8;
|
|
let d: utf8.decoder = utf8.decode(src[0:3]);
|
|
match (utf8.next(&d)) {
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
case let r: utf8.invalid => { fail(); };
|
|
case let r: rune => { if (r != 0x20ACu32: rune) { fail(); }; };
|
|
};
|
|
};
|
|
|
|
@test fn decode_four_byte() void = {
|
|
let src: [4]u8;
|
|
src[0] = 0xF0u8; src[1] = 0x9Fu8; src[2] = 0xA6u8; src[3] = 0x80u8;
|
|
let d: utf8.decoder = utf8.decode(src[0:4]);
|
|
match (utf8.next(&d)) {
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
case let r: utf8.invalid => { fail(); };
|
|
case let r: rune => { if (r != 0x1F980u32: rune) { fail(); }; };
|
|
};
|
|
};
|
|
|
|
// ---- decoder.next: invalid inputs -------------------------------------
|
|
// Cases mirror ref/hare/encoding/utf8/decode.ha:127-167 (surrogate /
|
|
// overlong / out-of-range / bad-continuation).
|
|
|
|
@test fn decode_surrogate() void = {
|
|
let src: [3]u8;
|
|
src[0] = 0xEDu8; src[1] = 0xA0u8; src[2] = 0x80u8; // U+D800
|
|
let d: utf8.decoder = utf8.decode(src[0:3]);
|
|
match (utf8.next(&d)) {
|
|
case let r: utf8.invalid => void;
|
|
case let r: rune => { fail(); };
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
};
|
|
};
|
|
|
|
@test fn decode_overlong() void = {
|
|
let src: [4]u8;
|
|
src[0] = 0xF0u8; src[1] = 0x82u8; src[2] = 0x82u8; src[3] = 0xACu8;
|
|
let d: utf8.decoder = utf8.decode(src[0:4]);
|
|
match (utf8.next(&d)) {
|
|
case let r: utf8.invalid => void;
|
|
case let r: rune => { fail(); };
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
};
|
|
};
|
|
|
|
@test fn decode_out_of_range() void = {
|
|
let src: [4]u8;
|
|
src[0] = 0xF5u8; src[1] = 0x94u8; src[2] = 0x80u8; src[3] = 0x80u8;
|
|
let d: utf8.decoder = utf8.decode(src[0:4]);
|
|
match (utf8.next(&d)) {
|
|
case let r: utf8.invalid => void;
|
|
case let r: rune => { fail(); };
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
};
|
|
};
|
|
|
|
// ref/hare/encoding/utf8/decode.ha:141 — `[0xC2, 0xFF]`: legal 2-byte
|
|
// lead followed by non-continuation. Pins next-table cell (state 1,
|
|
// byte 0xFF) returning -1; distinct from overlong (which is filtered
|
|
// in state 3/5/7 by lead-byte-aware sub-states).
|
|
@test fn decode_bad_continuation() void = {
|
|
let src: [2]u8;
|
|
src[0] = 0xC2u8; src[1] = 0xFFu8;
|
|
let d: utf8.decoder = utf8.decode(src[0:2]);
|
|
match (utf8.next(&d)) {
|
|
case let r: utf8.invalid => void;
|
|
case let r: rune => { fail(); };
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
};
|
|
};
|
|
|
|
// ref/hare/encoding/utf8/decode.ha:151 — `[0xF4, 0x8F, 0xBF, 0xBF]` =
|
|
// U+10FFFF, the largest legal codepoint. Pins the upper boundary;
|
|
// pairs with the existing `0xF5…` out-of-range row.
|
|
@test fn decode_max_in_range() void = {
|
|
let src: [4]u8;
|
|
src[0] = 0xF4u8; src[1] = 0x8Fu8; src[2] = 0xBFu8; src[3] = 0xBFu8;
|
|
let d: utf8.decoder = utf8.decode(src[0:4]);
|
|
match (utf8.next(&d)) {
|
|
case let r: rune => { if (r != 0x10FFFFu32: rune) { fail(); }; };
|
|
case let r: utf8.invalid => { fail(); };
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
};
|
|
};
|
|
|
|
// ---- decoder.next: truncated → more -----------------------------------
|
|
|
|
@test fn decode_truncated() void = {
|
|
let src: [2]u8;
|
|
src[0] = 0xE3u8; src[1] = 0x81u8; // missing 3rd byte
|
|
let d: utf8.decoder = utf8.decode(src[0:2]);
|
|
match (utf8.next(&d)) {
|
|
case utf8.more => void;
|
|
case let r: rune => { fail(); };
|
|
case utf8.done => { fail(); };
|
|
case let r: utf8.invalid => { fail(); };
|
|
};
|
|
};
|
|
|
|
// ---- decoder.next: done at EOI ----------------------------------------
|
|
|
|
@test fn decode_done() void = {
|
|
let src: [1]u8;
|
|
let d: utf8.decoder = utf8.decode(src[0:0]);
|
|
match (utf8.next(&d)) {
|
|
case utf8.done => void;
|
|
case let r: rune => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
case let r: utf8.invalid => { fail(); };
|
|
};
|
|
};
|
|
|
|
// ---- validate: well-formed mixed-width vs malformed -------------------
|
|
// ref/hare/encoding/utf8/decode.ha:207. Mixed input mirrors Hare's
|
|
// decode @test ('こんにちは' + NUL).
|
|
|
|
@test fn validate_mixed_ok() void = {
|
|
let src: [16]u8;
|
|
src[0] = 0xE3u8; src[1] = 0x81u8; src[2] = 0x93u8; // こ
|
|
src[3] = 0xE3u8; src[4] = 0x82u8; src[5] = 0x93u8; // ん
|
|
src[6] = 0xE3u8; src[7] = 0x81u8; src[8] = 0xABu8; // に
|
|
src[9] = 0xE3u8; src[10] = 0x81u8; src[11] = 0xA1u8; // ち
|
|
src[12] = 0xE3u8; src[13] = 0x81u8; src[14] = 0xAFu8; // は
|
|
src[15] = 0u8;
|
|
match (utf8.validate(src[0:16])) {
|
|
case let e: utf8.invalid => { fail(); };
|
|
case void => void;
|
|
};
|
|
};
|
|
|
|
@test fn validate_malformed() void = {
|
|
let src: [3]u8;
|
|
src[0] = 0xEDu8; src[1] = 0xA0u8; src[2] = 0x80u8; // surrogate
|
|
match (utf8.validate(src[0:3])) {
|
|
case let e: utf8.invalid => void;
|
|
case void => { fail(); };
|
|
};
|
|
};
|
|
|
|
@test fn validate_empty_ok() void = {
|
|
let src: [1]u8;
|
|
match (utf8.validate(src[0:0])) {
|
|
case let e: utf8.invalid => { fail(); };
|
|
case void => void;
|
|
};
|
|
};
|
|
|
|
// ---- round-trip: encode → decode → equal rune --------------------------
|
|
|
|
@test fn roundtrip() void = {
|
|
let runes: [4]u32;
|
|
runes[0] = 0x41u32;
|
|
runes[1] = 0xE9u32;
|
|
runes[2] = 0x20ACu32;
|
|
runes[3] = 0x1F980u32;
|
|
let i: i32 = 0;
|
|
for (i < 4) {
|
|
let buf: [4]u8;
|
|
let n: i32 = utf8.encoderune(buf[0:4], runes[i]: rune);
|
|
let d: utf8.decoder = utf8.decode(buf[0:n]);
|
|
match (utf8.next(&d)) {
|
|
case let r: rune => {
|
|
if ((r: u32) != runes[i]) { fail(); };
|
|
};
|
|
case utf8.done => { fail(); };
|
|
case utf8.more => { fail(); };
|
|
case let e: utf8.invalid => { fail(); };
|
|
};
|
|
i += 1;
|
|
};
|
|
};
|
|
|
|
export fn main() i32 = {
|
|
signalled = 1; runesz_ranges();
|
|
signalled = 2; utf8sz_classify();
|
|
signalled = 3; encode_ascii();
|
|
signalled = 4; encode_two_byte();
|
|
signalled = 5; encode_three_byte();
|
|
signalled = 6; encode_four_byte();
|
|
signalled = 7; decode_one_byte();
|
|
signalled = 8; decode_two_byte();
|
|
signalled = 9; decode_three_byte();
|
|
signalled = 10; decode_four_byte();
|
|
signalled = 11; decode_surrogate();
|
|
signalled = 12; decode_overlong();
|
|
signalled = 13; decode_out_of_range();
|
|
signalled = 14; decode_bad_continuation();
|
|
signalled = 15; decode_max_in_range();
|
|
signalled = 16; decode_truncated();
|
|
signalled = 17; decode_done();
|
|
signalled = 18; validate_mixed_ok();
|
|
signalled = 19; validate_malformed();
|
|
signalled = 20; validate_empty_ok();
|
|
signalled = 21; roundtrip();
|
|
return 0;
|
|
};
|