Replace the cmd/ww + selfhost driver's file-walk import resolver with true directory enumeration. `import encoding.utf8;` now finds the lib/encoding/utf8/ directory and concatenates every *.ww file in it (excluding *test.ww and the driver's *.combined.ww artifacts) in byte-wise sorted order, instead of just finding the single lib/encoding/utf8/utf8.ww file. Mirrors Hare's hare/module/srcs.ha:183 _findsrcs minus tag handling. Lookup order in both stages: (1) <dir>/<dot-as-slash>/ as directory → enumerate. (2) <dir>/<dot-as-slash>.ww as file. The legacy <dir>/<name>/<name>.ww shape from #18's retained divergence is dropped per rule-9 Hare-fidelity — Hare has no foo/foo.ha fallback; a module IS the directory. Symmetric across cstage (cmd/ww/main.c via opendir+qsort+stat) and wwstage (selfhost/cmd/ww/main.ww via existing lib/os.getdents64 + os.stat — no new lib/os surface needed; the rundirtests() walker in main.ww from #18 was the model). Bootstrap ww2.s==ww3.s==ww4.s byte-identical post-change. Bundling justification (rule 11): strict-same-package validation is bundled because the failure mode is dir-enum's own (a non-dir-enum compilation unit cannot trigger mismatch across enumerated files). The natural enforcement site is the driver — the parser can't distinguish dir-enum concat from file-walk concat. Both stages peek each file's first `package <name>;` line in expand_dir / expanddir and exit(1) on mismatch with a precise error pointing at the offending file. Hare's hare/module/srcs.ha:131 has the same constraint via its README gate. Other half of #23 (strict missing-package error tightening — 63 inline-source test wrappers blocker) stays deferred per its filing. Parser side (cmd/wcc/parse.c parseuse + lib/ww/parse/decl.ww parseuse): n->str now carries only the LEAF identifier from a dotted import. With the driver translating the full dotted path to a directory walk, the checker only needs the package bareword (last component) for the N_USE → decl disambiguation walk in check.c's src_imports / decl_mod. Mirrors Hare's `use encoding::utf8;` → `utf8::name` semantics (ref/hare/hare/ast/import.ha:7). Migration: lib/ww/sym.ww drops `import typ; import ast;`; lib/ww/parse/parse.ww drops `import expr; import stmt; import decl;`; lib/ww/lex/lex.ww drops `import tok;` — all sibling imports auto-resolve via the new dir-enum when callers import the package directory. lib/strings/, lib/encoding/utf8/utf8test.ww migrate `import utf8;` → `import encoding.utf8;`. Makefile drops -I lib/encoding/utf8 stopgap from wwdump_ww + w6c_ww. Seven test wrappers (700_e2e, 966_strings_run, 970_fmt_run, 971_log_run, 972_fnmatch_run, 982_getopt_run, 990_selfhost) and 995_self_rebuild drop the -I lib/encoding/utf8 runtime stopgap. Tests: new 737_direnum C wrapper + test/wcc/data/direnum/ fixtures pin (a) cross-pkg multi-file dir-enum build at runtime (both stages must succeed) and (b) strict-same-package mismatch error (both stages must surface "differs from" + exit non-zero). 738_module_decl gains row 6 pinning the n_use->str leaf-only storage post-parser change. Retained workaround at selfhost/cmd/ww/main.ww expanddir loop: `names[i][k]` nested-deref-then-index split into `let nm: *u8 = names[i]; nm[k]` because wwstage cgen miscompiles the chained form (treats inner u8 element as 8B sizeof *u8 instead of 1B sizeof u8: extra MOVQ $8 + IMULQ on the inner index, MOVQ instead of MOVZBQ load). Inline rule-8 WHY comment cites task #24 (wwstage cgen chained-index inner element size on **T). Two-step form routes through the bare-pointer index path which both stages handle byte-identically. Class A wwstage cgen UNDER (chained-index inner element size on **T) surfaced first time the codebase exercises the **T[i][k] shape via enumeratedir() — corpus-coverage-blind landmine pattern, same family as the trio (#27/#28/#31) from STATUS-5. 112/112 ok. ww2 == ww3 == ww4 byte-id holds.
424 lines
14 KiB
Plaintext
424 lines
14 KiB
Plaintext
// stringstest — exercises lib/strings. Run with
|
||
// `out/bin/ww run lib/strings/stringstest.ww`.
|
||
// Same signalled-then-fail()-with-+10 shape as bytes / utf8 / hex /
|
||
// time tests: non-zero exit pinpoints the failing scenario.
|
||
//
|
||
// Vectors mirror ref/hare/strings/{dup,concat,trim,contains,index,
|
||
// suffix,compare}.ha where ww can express them.
|
||
|
||
package strings;
|
||
|
||
import strings;
|
||
import encoding.utf8;
|
||
import os;
|
||
|
||
let signalled: i32 = 0;
|
||
fn fail() void = { os.exit(signalled + 10); };
|
||
|
||
fn streq(a: str, b: str) bool = {
|
||
if (a.len != b.len) { return false; };
|
||
let i: i32 = 0;
|
||
for (i < a.len) {
|
||
if (a[i] != b[i]) { return false; };
|
||
i += 1;
|
||
};
|
||
return true;
|
||
};
|
||
|
||
// ---- dup --------------------------------------------------------------
|
||
// ref/hare/strings/dup.ha:45.
|
||
|
||
@test fn dup_cases() void = {
|
||
let e: str = strings.dup("");
|
||
if (!streq(e, "")) { fail(); };
|
||
if (e.len != 0) { fail(); };
|
||
|
||
let h: str = strings.dup("hello");
|
||
if (!streq(h, "hello")) { fail(); };
|
||
defer os.free(h.ptr: *void, h.len: u64);
|
||
|
||
// multi-byte UTF-8: dup must copy raw bytes, not aliased view.
|
||
let m: str = strings.dup("こんにちは");
|
||
if (m.len != 15) { fail(); };
|
||
if (!streq(m, "こんにちは")) { fail(); };
|
||
if (m.ptr == "こんにちは".ptr) { fail(); }; // fresh alloc
|
||
defer os.free(m.ptr: *void, m.len: u64);
|
||
};
|
||
|
||
// ---- concat -----------------------------------------------------------
|
||
// ref/hare/strings/concat.ha:18 (2-arg subset).
|
||
|
||
@test fn concat_cases() void = {
|
||
let a: str = strings.concat("hello ", "world");
|
||
if (!streq(a, "hello world")) { fail(); };
|
||
defer os.free(a.ptr: *void, a.len: u64);
|
||
|
||
let e: str = strings.concat("", "");
|
||
if (!streq(e, "")) { fail(); };
|
||
// e.len == 0 — os.free guarded, skip.
|
||
|
||
let l: str = strings.concat("", "world");
|
||
if (!streq(l, "world")) { fail(); };
|
||
defer os.free(l.ptr: *void, l.len: u64);
|
||
|
||
let r: str = strings.concat("hello", "");
|
||
if (!streq(r, "hello")) { fail(); };
|
||
defer os.free(r.ptr: *void, r.len: u64);
|
||
|
||
let m: str = strings.concat("こん", "にちは");
|
||
if (!streq(m, "こんにちは")) { fail(); };
|
||
defer os.free(m.ptr: *void, m.len: u64);
|
||
};
|
||
|
||
// ---- hasprefix --------------------------------------------------------
|
||
// ref/hare/strings/suffix.ha:18.
|
||
|
||
@test fn hasprefix_cases() void = {
|
||
if (!strings.hasprefix("hello world", "hello")) { fail(); };
|
||
if (!strings.hasprefix("hello world", 'h')) { fail(); };
|
||
if ( strings.hasprefix("hello world", "world")) { fail(); };
|
||
if ( strings.hasprefix("hello world", 'q')) { fail(); };
|
||
if (!strings.hasprefix("hello", "hello")) { fail(); }; // equal-len
|
||
if (!strings.hasprefix("anything", "")) { fail(); }; // empty prefix
|
||
if ( strings.hasprefix("", "x")) { fail(); };
|
||
// multibyte rune prefix — '\'é\'' literal blocked by single-byte
|
||
// lexrune (lib/ww/lex/lex.ww:659); pass codepoint directly.
|
||
if (!strings.hasprefix("éclat", 0xE9u32: rune)) { fail(); };
|
||
if (!strings.hasprefix("🦀rust", 0x1F980u32: rune)) { fail(); };
|
||
};
|
||
|
||
// ---- hassuffix --------------------------------------------------------
|
||
// ref/hare/strings/suffix.ha:36.
|
||
|
||
@test fn hassuffix_cases() void = {
|
||
if (!strings.hassuffix("hello world", "world")) { fail(); };
|
||
if (!strings.hassuffix("hello world", 'd')) { fail(); };
|
||
if ( strings.hassuffix("hello world", "hello")) { fail(); };
|
||
if ( strings.hassuffix("hello world", 'h')) { fail(); };
|
||
if (!strings.hassuffix("café", 0xE9u32: rune)) { fail(); }; // multibyte
|
||
};
|
||
|
||
// ---- contains ---------------------------------------------------------
|
||
// ref/hare/strings/contains.ha:27.
|
||
|
||
@test fn contains_cases() void = {
|
||
if (!strings.contains("hello world", "hello")) { fail(); };
|
||
if (!strings.contains("hello world", 'h')) { fail(); };
|
||
if ( strings.contains("hello world", 'x')) { fail(); };
|
||
if (!strings.contains("hello world", "world")) { fail(); };
|
||
if (!strings.contains("hello world", "")) { fail(); }; // empty hits at 0
|
||
if ( strings.contains("hello world", "foobar")) { fail(); };
|
||
if (!strings.contains("こんにちは", 0x306Bu32: rune)) { fail(); }; // 'に'
|
||
if (!strings.contains("こんにちは", "ちは")) { fail(); };
|
||
};
|
||
|
||
// ---- byteindex --------------------------------------------------------
|
||
// ref/hare/strings/index.ha:147 (byteindex tests, both arms).
|
||
|
||
@test fn byteindex_str_cases() void = {
|
||
match (strings.byteindex("hello", "hello")) {
|
||
case let i: i32 => { if (i != 0) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
match (strings.byteindex("hello world!", "world")) {
|
||
case let i: i32 => { if (i != 6) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
match (strings.byteindex("hello world!", "orld!")) {
|
||
case let i: i32 => { if (i != 7) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
match (strings.byteindex("hello world!", "word")) {
|
||
case let i: i32 => { fail(); };
|
||
case void => void;
|
||
};
|
||
// empty needle hits at 0 (ref/hare/bytes/index.ha:63).
|
||
match (strings.byteindex("hello", "")) {
|
||
case let i: i32 => { if (i != 0) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// empty haystack, non-empty needle — absent.
|
||
match (strings.byteindex("", "x")) {
|
||
case let i: i32 => { fail(); };
|
||
case void => void;
|
||
};
|
||
// multibyte substring in multibyte haystack.
|
||
match (strings.byteindex("こんにちは", "ちは")) {
|
||
case let i: i32 => { if (i != 9) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
};
|
||
|
||
@test fn byteindex_rune_cases() void = {
|
||
// ASCII rune (1-byte encoding).
|
||
match (strings.byteindex("hello world", 'w')) {
|
||
case let i: i32 => { if (i != 6) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// 2-byte rune U+00E9 'é' inside "café".
|
||
match (strings.byteindex("café", 0xE9u32: rune)) {
|
||
case let i: i32 => { if (i != 3) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// 3-byte rune U+3061 'ち' inside "こんにちは".
|
||
match (strings.byteindex("こんにちは", 0x3061u32: rune)) {
|
||
case let i: i32 => { if (i != 9) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// 4-byte rune U+1F980 '🦀' inside "ab🦀cd".
|
||
match (strings.byteindex("ab🦀cd", 0x1F980u32: rune)) {
|
||
case let i: i32 => { if (i != 2) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// absent.
|
||
match (strings.byteindex("こんにちは", 'q')) {
|
||
case let i: i32 => { fail(); };
|
||
case void => void;
|
||
};
|
||
};
|
||
|
||
// ---- rbyteindex -------------------------------------------------------
|
||
|
||
@test fn rbyteindex_cases() void = {
|
||
// Two 'た' in "またあったね" — ref/hare/strings/index.ha:160-161.
|
||
match (strings.byteindex("またあったね", "た")) {
|
||
case let i: i32 => { if (i != 3) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
match (strings.rbyteindex("またあったね", "た")) {
|
||
case let i: i32 => { if (i != 12) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// Rune arm, multi-byte 'に' U+306B.
|
||
match (strings.rbyteindex("こんにちは", 0x306Bu32: rune)) {
|
||
case let i: i32 => { if (i != 6) { fail(); }; };
|
||
case void => { fail(); };
|
||
};
|
||
// Absent.
|
||
match (strings.rbyteindex("abc", 'z')) {
|
||
case let i: i32 => { fail(); };
|
||
case void => void;
|
||
};
|
||
};
|
||
|
||
// ---- trimprefix / trimsuffix ------------------------------------------
|
||
// ref/hare/strings/trim.ha:99-107.
|
||
|
||
@test fn trimprefix_cases() void = {
|
||
if (!streq(strings.trimprefix("", ""), "")) { fail(); };
|
||
if (!streq(strings.trimprefix("", "blablabla"), "")) { fail(); };
|
||
if (!streq(strings.trimprefix("hello, world", "hello"), ", world")) { fail(); };
|
||
if (!streq(strings.trimprefix("blablabla", "bla"), "blabla")) { fail(); };
|
||
// equal-length match strips to empty.
|
||
if (!streq(strings.trimprefix("hello", "hello"), "")) { fail(); };
|
||
};
|
||
|
||
@test fn trimsuffix_cases() void = {
|
||
if (!streq(strings.trimsuffix("", ""), "")) { fail(); };
|
||
if (!streq(strings.trimsuffix("", "blablabla"), "")) { fail(); };
|
||
if (!streq(strings.trimsuffix("hello, world", "world"), "hello, ")) { fail(); };
|
||
if (!streq(strings.trimsuffix("blablabla", "bla"), "blabla")) { fail(); };
|
||
if (!streq(strings.trimsuffix("hello", "hello"), "")) { fail(); };
|
||
};
|
||
|
||
// ---- ltrim / rtrim / trim (single-rune subset) ------------------------
|
||
// ref/hare/strings/trim.ha:75-97. Vectors restricted to single-rune
|
||
// patterns (Hare's `rune...` blocks on task #16).
|
||
|
||
@test fn ltrim_cases() void = {
|
||
if (!streq(strings.ltrim("", 'x'), "")) { fail(); };
|
||
if (!streq(strings.ltrim("aaabc", 'a'), "bc")) { fail(); };
|
||
if (!streq(strings.ltrim("xyz", 'a'), "xyz")) { fail(); }; // no match
|
||
if (!streq(strings.ltrim("aaaa", 'a'), "")) { fail(); }; // all stripped
|
||
// 4-byte rune pattern — '𝚊' = U+1D68A.
|
||
if (!streq(strings.ltrim("𝚊𝚊hi", 0x1D68Au32: rune), "hi")) { fail(); };
|
||
};
|
||
|
||
@test fn rtrim_cases() void = {
|
||
if (!streq(strings.rtrim("", 'x'), "")) { fail(); };
|
||
if (!streq(strings.rtrim("bcaaa", 'a'), "bc")) { fail(); };
|
||
if (!streq(strings.rtrim("xyz", 'a'), "xyz")) { fail(); };
|
||
if (!streq(strings.rtrim("aaaa", 'a'), "")) { fail(); };
|
||
if (!streq(strings.rtrim("hi𝚊𝚊", 0x1D68Au32: rune), "hi")) { fail(); };
|
||
};
|
||
|
||
@test fn trim_cases() void = {
|
||
if (!streq(strings.trim("", 'x'), "")) { fail(); };
|
||
if (!streq(strings.trim("aaabcaaa", 'a'), "bc")) { fail(); };
|
||
if (!streq(strings.trim("xyz", 'a'), "xyz")) { fail(); };
|
||
if (!streq(strings.trim("aaaa", 'a'), "")) { fail(); };
|
||
};
|
||
|
||
// ---- compare ----------------------------------------------------------
|
||
// ref/hare/strings/compare.ha:16.
|
||
|
||
@test fn compare_cases() void = {
|
||
if (strings.compare("ABC", "ABC") != 0) { fail(); };
|
||
if (strings.compare("ABC", "AB") <= 0) { fail(); };
|
||
if (strings.compare("AB", "ABC") >= 0) { fail(); };
|
||
if (strings.compare("BCD", "ABC") <= 0) { fail(); };
|
||
if (strings.compare("ABC", "abc") >= 0) { fail(); };
|
||
};
|
||
|
||
// ---- toutf8 / fromutf8_unsafe roundtrip -------------------------------
|
||
// ref/hare/strings/utf8.ha:31.
|
||
|
||
@test fn utf8_roundtrip_cases() void = {
|
||
let s: str = "hello";
|
||
let b: []u8 = strings.toutf8(s);
|
||
if (b.len != 5) { fail(); };
|
||
if (b[0] != 104u8) { fail(); }; // 'h'
|
||
let r: str = strings.fromutf8_unsafe(b);
|
||
if (!streq(r, "hello")) { fail(); };
|
||
if (r.ptr != s.ptr) { fail(); }; // borrowed, not copied
|
||
};
|
||
|
||
// ---- iter / next ------------------------------------------------------
|
||
// ref/hare/strings/iter.ha:84-108. Hare's @test fn iter() uses prev +
|
||
// riter heavily; both are deferred (no `utf8.prev`). Rebuild forward-
|
||
// only here: empty / ASCII / 2-byte / 3-byte / 4-byte / done@EOI /
|
||
// mixed-width.
|
||
|
||
@test fn iter_empty_cases() void = {
|
||
let it: strings.iterator = strings.iter("");
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
// Repeated next after done stays done.
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
};
|
||
|
||
@test fn iter_ascii_cases() void = {
|
||
let it: strings.iterator = strings.iter("hi!");
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != 'h') { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != 'i') { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != '!') { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
};
|
||
|
||
@test fn iter_twobyte_cases() void = {
|
||
let it: strings.iterator = strings.iter("café");
|
||
let i: i32 = 0;
|
||
let expect: [4]rune;
|
||
expect[0] = 'c'; expect[1] = 'a'; expect[2] = 'f';
|
||
expect[3] = 0xE9u32: rune; // 'é' U+00E9
|
||
for (i < 4) {
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != expect[i]) { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
i += 1;
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
};
|
||
|
||
@test fn iter_threebyte_cases() void = {
|
||
let it: strings.iterator = strings.iter("こんにちは");
|
||
let i: i32 = 0;
|
||
let expect: [5]rune;
|
||
expect[0] = 0x3053u32: rune; // 'こ'
|
||
expect[1] = 0x3093u32: rune; // 'ん'
|
||
expect[2] = 0x306Bu32: rune; // 'に'
|
||
expect[3] = 0x3061u32: rune; // 'ち'
|
||
expect[4] = 0x306Fu32: rune; // 'は'
|
||
for (i < 5) {
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != expect[i]) { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
i += 1;
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
};
|
||
|
||
@test fn iter_fourbyte_cases() void = {
|
||
let it: strings.iterator = strings.iter("🦀rust");
|
||
let i: i32 = 0;
|
||
let expect: [5]rune;
|
||
expect[0] = 0x1F980u32: rune; // '🦀'
|
||
expect[1] = 'r'; expect[2] = 'u'; expect[3] = 's'; expect[4] = 't';
|
||
for (i < 5) {
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != expect[i]) { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
i += 1;
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
};
|
||
|
||
@test fn iter_mixed_cases() void = {
|
||
// "Hello, 世界! 🌍" — 1+1+1+1+1+1+1+3+3+1+1+4 = 12 runes,
|
||
// widths 1/3/4 mixed.
|
||
let it: strings.iterator = strings.iter("Hello, 世界! 🌍");
|
||
let i: i32 = 0;
|
||
let expect: [12]rune;
|
||
expect[0] = 'H'; expect[1] = 'e'; expect[2] = 'l'; expect[3] = 'l';
|
||
expect[4] = 'o'; expect[5] = ','; expect[6] = ' ';
|
||
expect[7] = 0x4E16u32: rune; // '世'
|
||
expect[8] = 0x754Cu32: rune; // '界'
|
||
expect[9] = '!'; expect[10] = ' ';
|
||
expect[11] = 0x1F30Du32: rune; // '🌍'
|
||
for (i < 12) {
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { if (r != expect[i]) { fail(); }; };
|
||
case utf8.done => { fail(); };
|
||
};
|
||
i += 1;
|
||
};
|
||
match (strings.next(&it)) {
|
||
case let r: rune => { fail(); };
|
||
case utf8.done => void;
|
||
};
|
||
};
|
||
|
||
export fn main() i32 = {
|
||
signalled = 1; dup_cases();
|
||
signalled = 2; concat_cases();
|
||
signalled = 3; hasprefix_cases();
|
||
signalled = 4; hassuffix_cases();
|
||
signalled = 5; contains_cases();
|
||
signalled = 6; byteindex_str_cases();
|
||
signalled = 7; byteindex_rune_cases();
|
||
signalled = 8; rbyteindex_cases();
|
||
signalled = 9; trimprefix_cases();
|
||
signalled = 10; trimsuffix_cases();
|
||
signalled = 11; ltrim_cases();
|
||
signalled = 12; rtrim_cases();
|
||
signalled = 13; trim_cases();
|
||
signalled = 14; compare_cases();
|
||
signalled = 15; utf8_roundtrip_cases();
|
||
signalled = 16; iter_empty_cases();
|
||
signalled = 17; iter_ascii_cases();
|
||
signalled = 18; iter_twobyte_cases();
|
||
signalled = 19; iter_threebyte_cases();
|
||
signalled = 20; iter_fourbyte_cases();
|
||
signalled = 21; iter_mixed_cases();
|
||
return 0;
|
||
};
|