Files
ww/lib/strings/stringstest.ww
Hojun-Cho d09197af8e lib/strings+test: Hare port (iterator + next)
Forward UTF-8 rune cursor per ref/hare/strings/iter.ha; iterator
flattens Hare's anon-embedded utf8::decoder to explicit
offs/src/reverse fields, next() copy-in/copy-out a local decoder
and aborts on more/invalid per move()'s discipline.

The iterator flattens Hare's anonymous-embedded utf8::decoder
(ref/hare/strings/iter.ha:6-9) to explicit offs/src/reverse
fields because ww has no anon-embed syntax. reverse is retained
on the struct so riter populates it once utf8.prev (reverse DFA)
and strings.prev land.

next() copies the iterator's offs/src into a local utf8.decoder,
delegates to utf8.next, then writes offs back; copy-in/copy-out
is the cost of the flattened layout. more/invalid arms abort with
"strings.next: invalid UTF-8", mirroring Hare's move()
(ref/hare/strings/iter.ha:51-58) which aborts unconditionally
on those arms.

Deferred surface (no in-tree caller; follow-up tasks): prev, riter,
iterstr, slice, position, move. prev specifically needs utf8.prev
(reverse DFA), which isn't on the lib/encoding/utf8 surface yet.

lib/strings/strings.ww moves off test/wcc/900_stdlib.c's
standalone-compile list per the existing bufio/fmt/os precedent:
the iterator's (rune | utf8.done) return type and the local
utf8.decoder reference need cross-module type resolution, which
the standalone w6c path doesn't do. Runtime coverage stays at
966_strings_run, which now exercises 21 signalled cases (was 15).

The iterator's Hare-faithful `reverse: bool` field surfaced #33
(cstage cgen narrow-sret-field copy mis-width) during this port's
pre-flight; that fix landed at 0d96196 ahead of this commit.
Future bisecters tracking a narrow-field-related regression in
lib/strings or sret-aware stdlib growth should consult #33 + this
commit's bracket. The 4-arm match on utf8.next return surfaced
#31 (wwstage fnretlookupmod) earlier in the same chain (b787641).

Tests:
  - 6 new @test fns in stringstest.ww (signalled 16-21): empty,
    ASCII, 2-byte (café), 3-byte (こんにちは), 4-byte (🦀rust),
    mixed-width ("Hello, 世界! 🌍"). Each verifies forward
    iteration, done@EOI, and repeated next-after-done stays done
    (iter_empty). Multibyte literal limitation handled via
    `0xE9u32: rune` cast per existing pattern at stringstest.ww:82.

104/104 ok. 995_self_rebuild stays green (ww2==ww3==ww4 byte-id).
2026-05-18 12:16:22 +09:00

422 lines
14 KiB
Plaintext
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// stringstest — exercises lib/strings. Run with
// `out/bin/ww run lib/strings/stringstest.ww -I lib/encoding/utf8`.
// Same signalled-then-fail()-with-+10 shape as bytes / utf8 / hex /
// time tests: non-zero exit pinpoints the failing scenario.
//
// Vectors mirror ref/hare/strings/{dup,concat,trim,contains,index,
// suffix,compare}.ha where ww can express them.
use strings;
use utf8;
use os;
let signalled: i32 = 0;
fn fail() void = { os.exit(signalled + 10); };
fn streq(a: str, b: str) bool = {
if (a.len != b.len) { return false; };
let i: i32 = 0;
for (i < a.len) {
if (a[i] != b[i]) { return false; };
i += 1;
};
return true;
};
// ---- dup --------------------------------------------------------------
// ref/hare/strings/dup.ha:45.
@test fn dup_cases() void = {
let e: str = strings.dup("");
if (!streq(e, "")) { fail(); };
if (e.len != 0) { fail(); };
let h: str = strings.dup("hello");
if (!streq(h, "hello")) { fail(); };
defer os.free(h.ptr: *void, h.len: u64);
// multi-byte UTF-8: dup must copy raw bytes, not aliased view.
let m: str = strings.dup("こんにちは");
if (m.len != 15) { fail(); };
if (!streq(m, "こんにちは")) { fail(); };
if (m.ptr == "こんにちは".ptr) { fail(); }; // fresh alloc
defer os.free(m.ptr: *void, m.len: u64);
};
// ---- concat -----------------------------------------------------------
// ref/hare/strings/concat.ha:18 (2-arg subset).
@test fn concat_cases() void = {
let a: str = strings.concat("hello ", "world");
if (!streq(a, "hello world")) { fail(); };
defer os.free(a.ptr: *void, a.len: u64);
let e: str = strings.concat("", "");
if (!streq(e, "")) { fail(); };
// e.len == 0 — os.free guarded, skip.
let l: str = strings.concat("", "world");
if (!streq(l, "world")) { fail(); };
defer os.free(l.ptr: *void, l.len: u64);
let r: str = strings.concat("hello", "");
if (!streq(r, "hello")) { fail(); };
defer os.free(r.ptr: *void, r.len: u64);
let m: str = strings.concat("こん", "にちは");
if (!streq(m, "こんにちは")) { fail(); };
defer os.free(m.ptr: *void, m.len: u64);
};
// ---- hasprefix --------------------------------------------------------
// ref/hare/strings/suffix.ha:18.
@test fn hasprefix_cases() void = {
if (!strings.hasprefix("hello world", "hello")) { fail(); };
if (!strings.hasprefix("hello world", 'h')) { fail(); };
if ( strings.hasprefix("hello world", "world")) { fail(); };
if ( strings.hasprefix("hello world", 'q')) { fail(); };
if (!strings.hasprefix("hello", "hello")) { fail(); }; // equal-len
if (!strings.hasprefix("anything", "")) { fail(); }; // empty prefix
if ( strings.hasprefix("", "x")) { fail(); };
// multibyte rune prefix — '\'é\'' literal blocked by single-byte
// lexrune (lib/ww/lex/lex.ww:659); pass codepoint directly.
if (!strings.hasprefix("éclat", 0xE9u32: rune)) { fail(); };
if (!strings.hasprefix("🦀rust", 0x1F980u32: rune)) { fail(); };
};
// ---- hassuffix --------------------------------------------------------
// ref/hare/strings/suffix.ha:36.
@test fn hassuffix_cases() void = {
if (!strings.hassuffix("hello world", "world")) { fail(); };
if (!strings.hassuffix("hello world", 'd')) { fail(); };
if ( strings.hassuffix("hello world", "hello")) { fail(); };
if ( strings.hassuffix("hello world", 'h')) { fail(); };
if (!strings.hassuffix("café", 0xE9u32: rune)) { fail(); }; // multibyte
};
// ---- contains ---------------------------------------------------------
// ref/hare/strings/contains.ha:27.
@test fn contains_cases() void = {
if (!strings.contains("hello world", "hello")) { fail(); };
if (!strings.contains("hello world", 'h')) { fail(); };
if ( strings.contains("hello world", 'x')) { fail(); };
if (!strings.contains("hello world", "world")) { fail(); };
if (!strings.contains("hello world", "")) { fail(); }; // empty hits at 0
if ( strings.contains("hello world", "foobar")) { fail(); };
if (!strings.contains("こんにちは", 0x306Bu32: rune)) { fail(); }; // 'に'
if (!strings.contains("こんにちは", "ちは")) { fail(); };
};
// ---- byteindex --------------------------------------------------------
// ref/hare/strings/index.ha:147 (byteindex tests, both arms).
@test fn byteindex_str_cases() void = {
match (strings.byteindex("hello", "hello")) {
case let i: i32 => { if (i != 0) { fail(); }; };
case void => { fail(); };
};
match (strings.byteindex("hello world!", "world")) {
case let i: i32 => { if (i != 6) { fail(); }; };
case void => { fail(); };
};
match (strings.byteindex("hello world!", "orld!")) {
case let i: i32 => { if (i != 7) { fail(); }; };
case void => { fail(); };
};
match (strings.byteindex("hello world!", "word")) {
case let i: i32 => { fail(); };
case void => void;
};
// empty needle hits at 0 (ref/hare/bytes/index.ha:63).
match (strings.byteindex("hello", "")) {
case let i: i32 => { if (i != 0) { fail(); }; };
case void => { fail(); };
};
// empty haystack, non-empty needle — absent.
match (strings.byteindex("", "x")) {
case let i: i32 => { fail(); };
case void => void;
};
// multibyte substring in multibyte haystack.
match (strings.byteindex("こんにちは", "ちは")) {
case let i: i32 => { if (i != 9) { fail(); }; };
case void => { fail(); };
};
};
@test fn byteindex_rune_cases() void = {
// ASCII rune (1-byte encoding).
match (strings.byteindex("hello world", 'w')) {
case let i: i32 => { if (i != 6) { fail(); }; };
case void => { fail(); };
};
// 2-byte rune U+00E9 'é' inside "café".
match (strings.byteindex("café", 0xE9u32: rune)) {
case let i: i32 => { if (i != 3) { fail(); }; };
case void => { fail(); };
};
// 3-byte rune U+3061 'ち' inside "こんにちは".
match (strings.byteindex("こんにちは", 0x3061u32: rune)) {
case let i: i32 => { if (i != 9) { fail(); }; };
case void => { fail(); };
};
// 4-byte rune U+1F980 '🦀' inside "ab🦀cd".
match (strings.byteindex("ab🦀cd", 0x1F980u32: rune)) {
case let i: i32 => { if (i != 2) { fail(); }; };
case void => { fail(); };
};
// absent.
match (strings.byteindex("こんにちは", 'q')) {
case let i: i32 => { fail(); };
case void => void;
};
};
// ---- rbyteindex -------------------------------------------------------
@test fn rbyteindex_cases() void = {
// Two 'た' in "またあったね" — ref/hare/strings/index.ha:160-161.
match (strings.byteindex("またあったね", "た")) {
case let i: i32 => { if (i != 3) { fail(); }; };
case void => { fail(); };
};
match (strings.rbyteindex("またあったね", "た")) {
case let i: i32 => { if (i != 12) { fail(); }; };
case void => { fail(); };
};
// Rune arm, multi-byte 'に' U+306B.
match (strings.rbyteindex("こんにちは", 0x306Bu32: rune)) {
case let i: i32 => { if (i != 6) { fail(); }; };
case void => { fail(); };
};
// Absent.
match (strings.rbyteindex("abc", 'z')) {
case let i: i32 => { fail(); };
case void => void;
};
};
// ---- trimprefix / trimsuffix ------------------------------------------
// ref/hare/strings/trim.ha:99-107.
@test fn trimprefix_cases() void = {
if (!streq(strings.trimprefix("", ""), "")) { fail(); };
if (!streq(strings.trimprefix("", "blablabla"), "")) { fail(); };
if (!streq(strings.trimprefix("hello, world", "hello"), ", world")) { fail(); };
if (!streq(strings.trimprefix("blablabla", "bla"), "blabla")) { fail(); };
// equal-length match strips to empty.
if (!streq(strings.trimprefix("hello", "hello"), "")) { fail(); };
};
@test fn trimsuffix_cases() void = {
if (!streq(strings.trimsuffix("", ""), "")) { fail(); };
if (!streq(strings.trimsuffix("", "blablabla"), "")) { fail(); };
if (!streq(strings.trimsuffix("hello, world", "world"), "hello, ")) { fail(); };
if (!streq(strings.trimsuffix("blablabla", "bla"), "blabla")) { fail(); };
if (!streq(strings.trimsuffix("hello", "hello"), "")) { fail(); };
};
// ---- ltrim / rtrim / trim (single-rune subset) ------------------------
// ref/hare/strings/trim.ha:75-97. Vectors restricted to single-rune
// patterns (Hare's `rune...` blocks on task #16).
@test fn ltrim_cases() void = {
if (!streq(strings.ltrim("", 'x'), "")) { fail(); };
if (!streq(strings.ltrim("aaabc", 'a'), "bc")) { fail(); };
if (!streq(strings.ltrim("xyz", 'a'), "xyz")) { fail(); }; // no match
if (!streq(strings.ltrim("aaaa", 'a'), "")) { fail(); }; // all stripped
// 4-byte rune pattern — '𝚊' = U+1D68A.
if (!streq(strings.ltrim("𝚊𝚊hi", 0x1D68Au32: rune), "hi")) { fail(); };
};
@test fn rtrim_cases() void = {
if (!streq(strings.rtrim("", 'x'), "")) { fail(); };
if (!streq(strings.rtrim("bcaaa", 'a'), "bc")) { fail(); };
if (!streq(strings.rtrim("xyz", 'a'), "xyz")) { fail(); };
if (!streq(strings.rtrim("aaaa", 'a'), "")) { fail(); };
if (!streq(strings.rtrim("hi𝚊𝚊", 0x1D68Au32: rune), "hi")) { fail(); };
};
@test fn trim_cases() void = {
if (!streq(strings.trim("", 'x'), "")) { fail(); };
if (!streq(strings.trim("aaabcaaa", 'a'), "bc")) { fail(); };
if (!streq(strings.trim("xyz", 'a'), "xyz")) { fail(); };
if (!streq(strings.trim("aaaa", 'a'), "")) { fail(); };
};
// ---- compare ----------------------------------------------------------
// ref/hare/strings/compare.ha:16.
@test fn compare_cases() void = {
if (strings.compare("ABC", "ABC") != 0) { fail(); };
if (strings.compare("ABC", "AB") <= 0) { fail(); };
if (strings.compare("AB", "ABC") >= 0) { fail(); };
if (strings.compare("BCD", "ABC") <= 0) { fail(); };
if (strings.compare("ABC", "abc") >= 0) { fail(); };
};
// ---- toutf8 / fromutf8_unsafe roundtrip -------------------------------
// ref/hare/strings/utf8.ha:31.
@test fn utf8_roundtrip_cases() void = {
let s: str = "hello";
let b: []u8 = strings.toutf8(s);
if (b.len != 5) { fail(); };
if (b[0] != 104u8) { fail(); }; // 'h'
let r: str = strings.fromutf8_unsafe(b);
if (!streq(r, "hello")) { fail(); };
if (r.ptr != s.ptr) { fail(); }; // borrowed, not copied
};
// ---- iter / next ------------------------------------------------------
// ref/hare/strings/iter.ha:84-108. Hare's @test fn iter() uses prev +
// riter heavily; both are deferred (no `utf8.prev`). Rebuild forward-
// only here: empty / ASCII / 2-byte / 3-byte / 4-byte / done@EOI /
// mixed-width.
@test fn iter_empty_cases() void = {
let it: strings.iterator = strings.iter("");
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
// Repeated next after done stays done.
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
};
@test fn iter_ascii_cases() void = {
let it: strings.iterator = strings.iter("hi!");
match (strings.next(&it)) {
case let r: rune => { if (r != 'h') { fail(); }; };
case utf8.done => { fail(); };
};
match (strings.next(&it)) {
case let r: rune => { if (r != 'i') { fail(); }; };
case utf8.done => { fail(); };
};
match (strings.next(&it)) {
case let r: rune => { if (r != '!') { fail(); }; };
case utf8.done => { fail(); };
};
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
};
@test fn iter_twobyte_cases() void = {
let it: strings.iterator = strings.iter("café");
let i: i32 = 0;
let expect: [4]rune;
expect[0] = 'c'; expect[1] = 'a'; expect[2] = 'f';
expect[3] = 0xE9u32: rune; // 'é' U+00E9
for (i < 4) {
match (strings.next(&it)) {
case let r: rune => { if (r != expect[i]) { fail(); }; };
case utf8.done => { fail(); };
};
i += 1;
};
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
};
@test fn iter_threebyte_cases() void = {
let it: strings.iterator = strings.iter("こんにちは");
let i: i32 = 0;
let expect: [5]rune;
expect[0] = 0x3053u32: rune; // 'こ'
expect[1] = 0x3093u32: rune; // 'ん'
expect[2] = 0x306Bu32: rune; // 'に'
expect[3] = 0x3061u32: rune; // 'ち'
expect[4] = 0x306Fu32: rune; // 'は'
for (i < 5) {
match (strings.next(&it)) {
case let r: rune => { if (r != expect[i]) { fail(); }; };
case utf8.done => { fail(); };
};
i += 1;
};
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
};
@test fn iter_fourbyte_cases() void = {
let it: strings.iterator = strings.iter("🦀rust");
let i: i32 = 0;
let expect: [5]rune;
expect[0] = 0x1F980u32: rune; // '🦀'
expect[1] = 'r'; expect[2] = 'u'; expect[3] = 's'; expect[4] = 't';
for (i < 5) {
match (strings.next(&it)) {
case let r: rune => { if (r != expect[i]) { fail(); }; };
case utf8.done => { fail(); };
};
i += 1;
};
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
};
@test fn iter_mixed_cases() void = {
// "Hello, 世界! 🌍" — 1+1+1+1+1+1+1+3+3+1+1+4 = 12 runes,
// widths 1/3/4 mixed.
let it: strings.iterator = strings.iter("Hello, 世界! 🌍");
let i: i32 = 0;
let expect: [12]rune;
expect[0] = 'H'; expect[1] = 'e'; expect[2] = 'l'; expect[3] = 'l';
expect[4] = 'o'; expect[5] = ','; expect[6] = ' ';
expect[7] = 0x4E16u32: rune; // '世'
expect[8] = 0x754Cu32: rune; // '界'
expect[9] = '!'; expect[10] = ' ';
expect[11] = 0x1F30Du32: rune; // '🌍'
for (i < 12) {
match (strings.next(&it)) {
case let r: rune => { if (r != expect[i]) { fail(); }; };
case utf8.done => { fail(); };
};
i += 1;
};
match (strings.next(&it)) {
case let r: rune => { fail(); };
case utf8.done => void;
};
};
export fn main() i32 = {
signalled = 1; dup_cases();
signalled = 2; concat_cases();
signalled = 3; hasprefix_cases();
signalled = 4; hassuffix_cases();
signalled = 5; contains_cases();
signalled = 6; byteindex_str_cases();
signalled = 7; byteindex_rune_cases();
signalled = 8; rbyteindex_cases();
signalled = 9; trimprefix_cases();
signalled = 10; trimsuffix_cases();
signalled = 11; ltrim_cases();
signalled = 12; rtrim_cases();
signalled = 13; trim_cases();
signalled = 14; compare_cases();
signalled = 15; utf8_roundtrip_cases();
signalled = 16; iter_empty_cases();
signalled = 17; iter_ascii_cases();
signalled = 18; iter_twobyte_cases();
signalled = 19; iter_threebyte_cases();
signalled = 20; iter_fourbyte_cases();
signalled = 21; iter_mixed_cases();
return 0;
};