From aa8d15182a4a0bdbe16ffc27b022990dbb5c0a94 Mon Sep 17 00:00:00 2001 From: Hojun-Cho Date: Mon, 18 May 2026 22:32:13 +0900 Subject: [PATCH] lib/strings+test: Hare port (prev + riter + iterstr + position) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port four of the five c3 strings functions per ref/hare/strings/iter.ha; strings.slice deferred behind task #34 (wwstage fnretlookup same-name cross-module phantom return-ABI fixup, sub-bug of #4e). - prev ref/hare/strings/iter.ha:49 - riter ref/hare/strings/iter.ha:32 - iterstr ref/hare/strings/iter.ha:63 - position ref/hare/strings/iter.ha:82 Also adds a private move() helper (ref/hare/strings/iter.ha:51) shared by next/prev. Hare's move picks the utf8 function via a fn-pointer (`let fun = if (forward) &utf8::next else &utf8::prev`); ww has no fn-pointers in scope yet, so move branches on `forward` and calls utf8.next or utf8.prev directly at each site. strings.next is updated to dispatch via move(!it.reverse, it) — c2's implementation always called utf8.next regardless of iter.reverse, which was correct for forward iter() but would walk forward on riter()-produced iterators too. With riter landed in this commit, next() now correctly walks backward on reverse iterators per Hare's ref/hare/strings/iter.ha:45. No in-tree consumer regression: only stringstest constructs iterators today. strings.iterstr uses ww's `[lo:hi]` slice syntax instead of Hare's `[lo..hi]`; same semantics (borrowed []u8 view). strings.slice deferred — Hare's body is `fromutf8_unsafe(utf8::slice(begin, end))` (ref/hare/strings/iter.ha:76). That delegation form triggers a wwstage fnretlookup miscompile when the caller module has a function of the same name as the callee (here: both strings::slice and utf8::slice exist), causing wwstage to emit the wrong return-ABI fixup (MOVQ DX, BX, str's AX/DX→AX/BX shim) after the cross-module CALL. Cstage handles the collision correctly; wwstage routes through the same-module function's return type and the byte-id checks (993_ww_ww + 995_self_rebuild) trip. Filed as task #34 with minimal repro; will land strings.slice when the fnretlookup graduate-by-module sub-bug in #34 is fixed. Top-of-file divergence note lists slice in the deferred set; the landing site keeps a comment-only stub. The public strings c3 surface ships 4-of-5 in this commit; slice + #34 land together in a follow-up. Tests: 5 new @test fns in stringstest.ww (signalled 22-26): iter_prev_at_start_cases (prev at offs=0 → done), iter_prev_ascii_cases (round-trip on forward iter), iter_full_cases (mirror of ref/hare/strings/iter.ha:84-108 — iter "こんにちは" with mid-walk iterstr + prev + next, then s = riter(...) sret-into-existing-slot for the reverse iterator pass), iter_position_cases (position tracks offs through a multibyte walk), iter_iterstr_reverse_cases (riter iterstr is bytes BEFORE the cursor, dual to forward iter's bytes-AFTER). The Hare @test fn iter body uses `s = riter("にちは")` mid-test to swap the iterator's direction (ref/hare/strings/iter.ha:101); ww's sret-into-existing-slot path handles that fine (probed pre-port). struct-copy let-from-ident (task #32) is sidestepped because no test creates a duplicate iterator via `let dup = it;`. 117/117 ok. 995_self_rebuild stays green (ww2==ww3==ww4 byte-id). 993_ww_ww + 994_w6c_ww also green (cstage/wwstage byte-identical on every corpus input including selfhost/cmd/wwdump/main.combined.ww). --- lib/strings/strings.ww | 107 ++++++++++++++++----- lib/strings/stringstest.ww | 133 +++++++++++++++++++++++++++ selfhost/cmd/w6c/main.combined.ww | 107 ++++++++++++++++----- selfhost/cmd/wwdump/main.combined.ww | 107 ++++++++++++++++----- selfhost/test/smoke.combined.ww | 107 ++++++++++++++++----- 5 files changed, 473 insertions(+), 88 deletions(-) diff --git a/lib/strings/strings.ww b/lib/strings/strings.ww index c26f2836..8bb66b83 100644 --- a/lib/strings/strings.ww +++ b/lib/strings/strings.ww @@ -25,10 +25,16 @@ // - `iterator` is flattened (`offs`, `src`, `reverse` fields). // Hare uses anonymous-embedded `utf8::decoder` // (ref/hare/strings/iter.ha:6-9); ww has no anonymous-embed -// syntax. `next` copies the iterator's `offs`/`src` into a local -// `utf8.decoder` for the call, then writes `offs` back. `prev` / -// `riter` / `iterstr` / `slice` / `position` are deferred — no -// in-tree caller; `prev` needs `utf8.prev` (reverse DFA). +// syntax, so `next`/`prev` copy `offs`/`src` into a local +// `utf8.decoder` for the call, then write `offs` back. +// - Hare's private `move()` helper dispatches on a `forward: bool` +// using a function-pointer `let fun = if (forward) &utf8::next +// else &utf8::prev`. ww has no fn-pointers in scope yet, so the +// dispatch is a branch on `forward` selecting the call site. +// - `slice` deferred — Hare's `fromutf8_unsafe(utf8::slice(begin, +// end))` (ref/hare/strings/iter.ha:76) hits wwstage fnretlookup +// same-name cross-module phantom return-ABI fixup (task #34). +// Lands once #34 clears. package strings; @@ -278,11 +284,11 @@ export fn trim(input: str, exclude: rune) str = { return ltrim(rtrim(input, exclude), exclude); }; -// iterator — forward UTF-8 rune cursor over a `str`. Layout flattens -// Hare's anonymous-embedded `utf8::decoder` -// (ref/hare/strings/iter.ha:6-9) to explicit fields; `reverse` is -// retained on the type because `riter` will populate it once `prev` / -// `utf8.prev` land. May be copied to save state. +// iterator — UTF-8 rune cursor over a `str`. Layout flattens Hare's +// anonymous-embedded `utf8::decoder` (ref/hare/strings/iter.ha:6-9) to +// explicit fields. `reverse` selects walk direction: forward iterators +// (`iter`) advance through utf8.next; reverse iterators (`riter`) advance +// through utf8.prev. May be copied to save state. export type iterator = struct { offs: i32, src: []u8, @@ -299,21 +305,78 @@ export fn iter(src: str) iterator = { return r; }; -// next — advance the iterator one rune. Returns `utf8.done` at end -// of input. Aborts on `more` / `invalid` — mirrors Hare's -// ref/hare/strings/iter.ha:51-58 `move()`, which aborts unconditionally -// on those arms ("Invalid UTF-8 string (this should not happen)"). -// -// Copy-in / copy-out is the cost of flattening the embedded decoder; -// see the iterator divergence note at the top of the file. -export fn next(it: *iterator) (rune | utf8.done) = { +// riter — initialize a reverse iterator at the end of `src`. `next` +// on a reverse iterator walks back through the string. +// ref/hare/strings/iter.ha:32. +export fn riter(src: str) iterator = { + let r: iterator; + r.src = toutf8(src); + r.offs = src.len; + r.reverse = true; + return r; +}; + +// move — private dispatch shared by next/prev. `forward` selects +// utf8.next vs utf8.prev. Aborts on more/invalid per Hare's +// ref/hare/strings/iter.ha:51-58 ("Invalid UTF-8 string (this should +// not happen)"). Hare picks the utf8 function via a fn-pointer; ww +// branches on `forward` at each call site instead. +fn move(forward: bool, it: *iterator) (rune | utf8.done) = { let d: utf8.decoder; d.src = it.src; d.offs = it.offs; - match (utf8.next(&d)) { - case let r: rune => { it.offs = d.offs; return r; }; - case let dn: utf8.done => return dn; - case let m: utf8.more => abort("strings.next: invalid UTF-8"); - case let e: utf8.invalid => abort("strings.next: invalid UTF-8"); + if (forward) { + match (utf8.next(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; + } else { + match (utf8.prev(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; }; }; + +// next — advance the iterator one rune. Forward iterators step +// through utf8.next; reverse iterators (riter) step backward through +// utf8.prev. Returns utf8.done at end-of-walk. ref/hare/strings/iter.ha:45. +export fn next(it: *iterator) (rune | utf8.done) = { + return move(!it.reverse, it); +}; + +// prev — step back one rune. Dual to next: on a forward iterator +// this walks utf8.prev; on a reverse iterator (riter) it walks +// utf8.next. ref/hare/strings/iter.ha:49. +export fn prev(it: *iterator) (rune | utf8.done) = { + return move(it.reverse, it); +}; + +// iterstr — borrowed view of the bytes remaining in the iterator's +// walk direction. Forward iter: bytes from offs to end; reverse iter: +// bytes from start to offs. ref/hare/strings/iter.ha:63. +export fn iterstr(it: *iterator) str = { + let r: []u8; + if (it.reverse) { + r = it.src[0:it.offs]; + } else { + r = it.src[it.offs:it.src.len]; + }; + return fromutf8_unsafe(r); +}; + +// strings.slice deferred — Hare's delegation form +// `fromutf8_unsafe(utf8::slice(begin, end))` (ref/hare/strings/iter.ha:76) +// triggers wwstage fnretlookup same-name cross-module phantom return-ABI +// fixup (task #34, Class A wwstage UNDER, sub-bug of #4e). Inlining the +// body would diverge from Hare at API level (rule 9). Land after #34. + +// position — byte-wise offset of the iterator in its source. +// ref/hare/strings/iter.ha:82. +export fn position(it: *iterator) i32 = { + return it.offs; +}; diff --git a/lib/strings/stringstest.ww b/lib/strings/stringstest.ww index 1d39574b..80d27840 100644 --- a/lib/strings/stringstest.ww +++ b/lib/strings/stringstest.ww @@ -397,6 +397,134 @@ fn streq(a: str, b: str) bool = { }; }; +// ---- prev / riter / iterstr / slice / position ----------------------- +// ref/hare/strings/iter.ha:84-127. The Hare @test fn iter body uses +// `s = riter(...)` mid-test to swap the iterator's direction; ww's +// sret-into-existing-slot path handles that fine (probed pre-port). + +@test fn iter_prev_at_start_cases() void = { + let it: strings.iterator = strings.iter("hi"); + match (strings.prev(&it)) { + case utf8.done => void; + case let r: rune => { fail(); }; + }; +}; + +@test fn iter_prev_ascii_cases() void = { + let it: strings.iterator = strings.iter("abc"); + match (strings.next(&it)) { + case let r: rune => { if (r != 'a') { fail(); }; }; + case utf8.done => { fail(); }; + }; + match (strings.prev(&it)) { + case let r: rune => { if (r != 'a') { fail(); }; }; + case utf8.done => { fail(); }; + }; + match (strings.prev(&it)) { + case utf8.done => void; + case let r: rune => { fail(); }; + }; +}; + +// Mirror of ref/hare/strings/iter.ha:84-108 — `iter("こんにちは")`, +// step+back+iterstr+riter-reassign sequence. +@test fn iter_full_cases() void = { + let s: strings.iterator = strings.iter("こんにちは"); + match (strings.prev(&s)) { + case utf8.done => void; + case let r: rune => { fail(); }; + }; + let expect1: [2]rune; + expect1[0] = 0x3053u32: rune; // 'こ' + expect1[1] = 0x3093u32: rune; // 'ん' + let i: i32 = 0; + for (i < 2) { + match (strings.next(&s)) { + case let r: rune => { if (r != expect1[i]) { fail(); }; }; + case utf8.done => { fail(); }; + }; + i += 1; + }; + if (!streq(strings.iterstr(&s), "にちは")) { fail(); }; + match (strings.prev(&s)) { + case let r: rune => { if (r != 0x3093u32: rune) { fail(); }; }; // 'ん' + case utf8.done => { fail(); }; + }; + let expect2: [4]rune; + expect2[0] = 0x3093u32: rune; // 'ん' + expect2[1] = 0x306Bu32: rune; // 'に' + expect2[2] = 0x3061u32: rune; // 'ち' + expect2[3] = 0x306Fu32: rune; // 'は' + i = 0; + for (i < 4) { + match (strings.next(&s)) { + case let r: rune => { if (r != expect2[i]) { fail(); }; }; + case utf8.done => { fail(); }; + }; + i += 1; + }; + match (strings.next(&s)) { + case utf8.done => void; + case let r: rune => { fail(); }; + }; + // Repeated next-after-done stays done. + match (strings.next(&s)) { + case utf8.done => void; + case let r: rune => { fail(); }; + }; + match (strings.prev(&s)) { + case let r: rune => { if (r != 0x306Fu32: rune) { fail(); }; }; // 'は' + case utf8.done => { fail(); }; + }; + + // Swap to a reverse iterator. sret-into-existing-slot. + s = strings.riter("にちは"); + let expect3: [3]rune; + expect3[0] = 0x306Fu32: rune; // 'は' + expect3[1] = 0x3061u32: rune; // 'ち' + expect3[2] = 0x306Bu32: rune; // 'に' + i = 0; + for (i < 3) { + match (strings.next(&s)) { + case let r: rune => { if (r != expect3[i]) { fail(); }; }; + case utf8.done => { fail(); }; + }; + i += 1; + }; + match (strings.next(&s)) { + case utf8.done => void; + case let r: rune => { fail(); }; + }; + match (strings.prev(&s)) { + case let r: rune => { if (r != 0x306Bu32: rune) { fail(); }; }; // 'に' + case utf8.done => { fail(); }; + }; +}; + +@test fn iter_position_cases() void = { + let it: strings.iterator = strings.iter("café"); // 5 bytes: c-a-f-é(2) + if (strings.position(&it) != 0) { fail(); }; + match (strings.next(&it)) { case let r: rune => void; case utf8.done => { fail(); }; }; + if (strings.position(&it) != 1) { fail(); }; + match (strings.next(&it)) { case let r: rune => void; case utf8.done => { fail(); }; }; + match (strings.next(&it)) { case let r: rune => void; case utf8.done => { fail(); }; }; + if (strings.position(&it) != 3) { fail(); }; + match (strings.next(&it)) { case let r: rune => void; case utf8.done => { fail(); }; }; + if (strings.position(&it) != 5) { fail(); }; +}; + +@test fn iter_iterstr_reverse_cases() void = { + // Reverse iter: iterstr is `src[0:offs]` — bytes BEFORE the cursor + // (the still-to-be-walked region in reverse direction). + let rit: strings.iterator = strings.riter("hello"); + if (!streq(strings.iterstr(&rit), "hello")) { fail(); }; + match (strings.next(&rit)) { case let r: rune => void; case utf8.done => { fail(); }; }; + if (!streq(strings.iterstr(&rit), "hell")) { fail(); }; + match (strings.next(&rit)) { case let r: rune => void; case utf8.done => { fail(); }; }; + match (strings.next(&rit)) { case let r: rune => void; case utf8.done => { fail(); }; }; + if (!streq(strings.iterstr(&rit), "he")) { fail(); }; +}; + export fn main() i32 = { signalled = 1; dup_cases(); signalled = 2; concat_cases(); @@ -419,5 +547,10 @@ export fn main() i32 = { signalled = 19; iter_threebyte_cases(); signalled = 20; iter_fourbyte_cases(); signalled = 21; iter_mixed_cases(); + signalled = 22; iter_prev_at_start_cases(); + signalled = 23; iter_prev_ascii_cases(); + signalled = 24; iter_full_cases(); + signalled = 25; iter_position_cases(); + signalled = 26; iter_iterstr_reverse_cases(); return 0; }; diff --git a/selfhost/cmd/w6c/main.combined.ww b/selfhost/cmd/w6c/main.combined.ww index 14db4f3b..9bdd9119 100644 --- a/selfhost/cmd/w6c/main.combined.ww +++ b/selfhost/cmd/w6c/main.combined.ww @@ -1467,10 +1467,16 @@ export fn position(d: *decoder) i32 = { // - `iterator` is flattened (`offs`, `src`, `reverse` fields). // Hare uses anonymous-embedded `utf8::decoder` // (ref/hare/strings/iter.ha:6-9); ww has no anonymous-embed -// syntax. `next` copies the iterator's `offs`/`src` into a local -// `utf8.decoder` for the call, then writes `offs` back. `prev` / -// `riter` / `iterstr` / `slice` / `position` are deferred — no -// in-tree caller; `prev` needs `utf8.prev` (reverse DFA). +// syntax, so `next`/`prev` copy `offs`/`src` into a local +// `utf8.decoder` for the call, then write `offs` back. +// - Hare's private `move()` helper dispatches on a `forward: bool` +// using a function-pointer `let fun = if (forward) &utf8::next +// else &utf8::prev`. ww has no fn-pointers in scope yet, so the +// dispatch is a branch on `forward` selecting the call site. +// - `slice` deferred — Hare's `fromutf8_unsafe(utf8::slice(begin, +// end))` (ref/hare/strings/iter.ha:76) hits wwstage fnretlookup +// same-name cross-module phantom return-ABI fixup (task #34). +// Lands once #34 clears. package strings; @@ -1720,11 +1726,11 @@ export fn trim(input: str, exclude: rune) str = { return ltrim(rtrim(input, exclude), exclude); }; -// iterator — forward UTF-8 rune cursor over a `str`. Layout flattens -// Hare's anonymous-embedded `utf8::decoder` -// (ref/hare/strings/iter.ha:6-9) to explicit fields; `reverse` is -// retained on the type because `riter` will populate it once `prev` / -// `utf8.prev` land. May be copied to save state. +// iterator — UTF-8 rune cursor over a `str`. Layout flattens Hare's +// anonymous-embedded `utf8::decoder` (ref/hare/strings/iter.ha:6-9) to +// explicit fields. `reverse` selects walk direction: forward iterators +// (`iter`) advance through utf8.next; reverse iterators (`riter`) advance +// through utf8.prev. May be copied to save state. export type iterator = struct { offs: i32, src: []u8, @@ -1741,25 +1747,82 @@ export fn iter(src: str) iterator = { return r; }; -// next — advance the iterator one rune. Returns `utf8.done` at end -// of input. Aborts on `more` / `invalid` — mirrors Hare's -// ref/hare/strings/iter.ha:51-58 `move()`, which aborts unconditionally -// on those arms ("Invalid UTF-8 string (this should not happen)"). -// -// Copy-in / copy-out is the cost of flattening the embedded decoder; -// see the iterator divergence note at the top of the file. -export fn next(it: *iterator) (rune | utf8.done) = { +// riter — initialize a reverse iterator at the end of `src`. `next` +// on a reverse iterator walks back through the string. +// ref/hare/strings/iter.ha:32. +export fn riter(src: str) iterator = { + let r: iterator; + r.src = toutf8(src); + r.offs = src.len; + r.reverse = true; + return r; +}; + +// move — private dispatch shared by next/prev. `forward` selects +// utf8.next vs utf8.prev. Aborts on more/invalid per Hare's +// ref/hare/strings/iter.ha:51-58 ("Invalid UTF-8 string (this should +// not happen)"). Hare picks the utf8 function via a fn-pointer; ww +// branches on `forward` at each call site instead. +fn move(forward: bool, it: *iterator) (rune | utf8.done) = { let d: utf8.decoder; d.src = it.src; d.offs = it.offs; - match (utf8.next(&d)) { - case let r: rune => { it.offs = d.offs; return r; }; - case let dn: utf8.done => return dn; - case let m: utf8.more => abort("strings.next: invalid UTF-8"); - case let e: utf8.invalid => abort("strings.next: invalid UTF-8"); + if (forward) { + match (utf8.next(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; + } else { + match (utf8.prev(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; }; }; +// next — advance the iterator one rune. Forward iterators step +// through utf8.next; reverse iterators (riter) step backward through +// utf8.prev. Returns utf8.done at end-of-walk. ref/hare/strings/iter.ha:45. +export fn next(it: *iterator) (rune | utf8.done) = { + return move(!it.reverse, it); +}; + +// prev — step back one rune. Dual to next: on a forward iterator +// this walks utf8.prev; on a reverse iterator (riter) it walks +// utf8.next. ref/hare/strings/iter.ha:49. +export fn prev(it: *iterator) (rune | utf8.done) = { + return move(it.reverse, it); +}; + +// iterstr — borrowed view of the bytes remaining in the iterator's +// walk direction. Forward iter: bytes from offs to end; reverse iter: +// bytes from start to offs. ref/hare/strings/iter.ha:63. +export fn iterstr(it: *iterator) str = { + let r: []u8; + if (it.reverse) { + r = it.src[0:it.offs]; + } else { + r = it.src[it.offs:it.src.len]; + }; + return fromutf8_unsafe(r); +}; + +// strings.slice deferred — Hare's delegation form +// `fromutf8_unsafe(utf8::slice(begin, end))` (ref/hare/strings/iter.ha:76) +// triggers wwstage fnretlookup same-name cross-module phantom return-ABI +// fixup (task #34, Class A wwstage UNDER, sub-bug of #4e). Inlining the +// body would diverge from Hare at API level (rule 9). Land after #34. + +// position — byte-wise offset of the iterator in its source. +// ref/hare/strings/iter.ha:82. +export fn position(it: *iterator) i32 = { + return it.offs; +}; + // strconv — number↔string conversions. // // Mirrors Hare's strconv:: surface. The *tos functions return a diff --git a/selfhost/cmd/wwdump/main.combined.ww b/selfhost/cmd/wwdump/main.combined.ww index d3e64751..972082db 100644 --- a/selfhost/cmd/wwdump/main.combined.ww +++ b/selfhost/cmd/wwdump/main.combined.ww @@ -1467,10 +1467,16 @@ export fn position(d: *decoder) i32 = { // - `iterator` is flattened (`offs`, `src`, `reverse` fields). // Hare uses anonymous-embedded `utf8::decoder` // (ref/hare/strings/iter.ha:6-9); ww has no anonymous-embed -// syntax. `next` copies the iterator's `offs`/`src` into a local -// `utf8.decoder` for the call, then writes `offs` back. `prev` / -// `riter` / `iterstr` / `slice` / `position` are deferred — no -// in-tree caller; `prev` needs `utf8.prev` (reverse DFA). +// syntax, so `next`/`prev` copy `offs`/`src` into a local +// `utf8.decoder` for the call, then write `offs` back. +// - Hare's private `move()` helper dispatches on a `forward: bool` +// using a function-pointer `let fun = if (forward) &utf8::next +// else &utf8::prev`. ww has no fn-pointers in scope yet, so the +// dispatch is a branch on `forward` selecting the call site. +// - `slice` deferred — Hare's `fromutf8_unsafe(utf8::slice(begin, +// end))` (ref/hare/strings/iter.ha:76) hits wwstage fnretlookup +// same-name cross-module phantom return-ABI fixup (task #34). +// Lands once #34 clears. package strings; @@ -1720,11 +1726,11 @@ export fn trim(input: str, exclude: rune) str = { return ltrim(rtrim(input, exclude), exclude); }; -// iterator — forward UTF-8 rune cursor over a `str`. Layout flattens -// Hare's anonymous-embedded `utf8::decoder` -// (ref/hare/strings/iter.ha:6-9) to explicit fields; `reverse` is -// retained on the type because `riter` will populate it once `prev` / -// `utf8.prev` land. May be copied to save state. +// iterator — UTF-8 rune cursor over a `str`. Layout flattens Hare's +// anonymous-embedded `utf8::decoder` (ref/hare/strings/iter.ha:6-9) to +// explicit fields. `reverse` selects walk direction: forward iterators +// (`iter`) advance through utf8.next; reverse iterators (`riter`) advance +// through utf8.prev. May be copied to save state. export type iterator = struct { offs: i32, src: []u8, @@ -1741,25 +1747,82 @@ export fn iter(src: str) iterator = { return r; }; -// next — advance the iterator one rune. Returns `utf8.done` at end -// of input. Aborts on `more` / `invalid` — mirrors Hare's -// ref/hare/strings/iter.ha:51-58 `move()`, which aborts unconditionally -// on those arms ("Invalid UTF-8 string (this should not happen)"). -// -// Copy-in / copy-out is the cost of flattening the embedded decoder; -// see the iterator divergence note at the top of the file. -export fn next(it: *iterator) (rune | utf8.done) = { +// riter — initialize a reverse iterator at the end of `src`. `next` +// on a reverse iterator walks back through the string. +// ref/hare/strings/iter.ha:32. +export fn riter(src: str) iterator = { + let r: iterator; + r.src = toutf8(src); + r.offs = src.len; + r.reverse = true; + return r; +}; + +// move — private dispatch shared by next/prev. `forward` selects +// utf8.next vs utf8.prev. Aborts on more/invalid per Hare's +// ref/hare/strings/iter.ha:51-58 ("Invalid UTF-8 string (this should +// not happen)"). Hare picks the utf8 function via a fn-pointer; ww +// branches on `forward` at each call site instead. +fn move(forward: bool, it: *iterator) (rune | utf8.done) = { let d: utf8.decoder; d.src = it.src; d.offs = it.offs; - match (utf8.next(&d)) { - case let r: rune => { it.offs = d.offs; return r; }; - case let dn: utf8.done => return dn; - case let m: utf8.more => abort("strings.next: invalid UTF-8"); - case let e: utf8.invalid => abort("strings.next: invalid UTF-8"); + if (forward) { + match (utf8.next(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; + } else { + match (utf8.prev(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; }; }; +// next — advance the iterator one rune. Forward iterators step +// through utf8.next; reverse iterators (riter) step backward through +// utf8.prev. Returns utf8.done at end-of-walk. ref/hare/strings/iter.ha:45. +export fn next(it: *iterator) (rune | utf8.done) = { + return move(!it.reverse, it); +}; + +// prev — step back one rune. Dual to next: on a forward iterator +// this walks utf8.prev; on a reverse iterator (riter) it walks +// utf8.next. ref/hare/strings/iter.ha:49. +export fn prev(it: *iterator) (rune | utf8.done) = { + return move(it.reverse, it); +}; + +// iterstr — borrowed view of the bytes remaining in the iterator's +// walk direction. Forward iter: bytes from offs to end; reverse iter: +// bytes from start to offs. ref/hare/strings/iter.ha:63. +export fn iterstr(it: *iterator) str = { + let r: []u8; + if (it.reverse) { + r = it.src[0:it.offs]; + } else { + r = it.src[it.offs:it.src.len]; + }; + return fromutf8_unsafe(r); +}; + +// strings.slice deferred — Hare's delegation form +// `fromutf8_unsafe(utf8::slice(begin, end))` (ref/hare/strings/iter.ha:76) +// triggers wwstage fnretlookup same-name cross-module phantom return-ABI +// fixup (task #34, Class A wwstage UNDER, sub-bug of #4e). Inlining the +// body would diverge from Hare at API level (rule 9). Land after #34. + +// position — byte-wise offset of the iterator in its source. +// ref/hare/strings/iter.ha:82. +export fn position(it: *iterator) i32 = { + return it.offs; +}; + // strconv — number↔string conversions. // // Mirrors Hare's strconv:: surface. The *tos functions return a diff --git a/selfhost/test/smoke.combined.ww b/selfhost/test/smoke.combined.ww index 355cd4e3..75e435b2 100644 --- a/selfhost/test/smoke.combined.ww +++ b/selfhost/test/smoke.combined.ww @@ -1358,10 +1358,16 @@ export fn position(d: *decoder) i32 = { // - `iterator` is flattened (`offs`, `src`, `reverse` fields). // Hare uses anonymous-embedded `utf8::decoder` // (ref/hare/strings/iter.ha:6-9); ww has no anonymous-embed -// syntax. `next` copies the iterator's `offs`/`src` into a local -// `utf8.decoder` for the call, then writes `offs` back. `prev` / -// `riter` / `iterstr` / `slice` / `position` are deferred — no -// in-tree caller; `prev` needs `utf8.prev` (reverse DFA). +// syntax, so `next`/`prev` copy `offs`/`src` into a local +// `utf8.decoder` for the call, then write `offs` back. +// - Hare's private `move()` helper dispatches on a `forward: bool` +// using a function-pointer `let fun = if (forward) &utf8::next +// else &utf8::prev`. ww has no fn-pointers in scope yet, so the +// dispatch is a branch on `forward` selecting the call site. +// - `slice` deferred — Hare's `fromutf8_unsafe(utf8::slice(begin, +// end))` (ref/hare/strings/iter.ha:76) hits wwstage fnretlookup +// same-name cross-module phantom return-ABI fixup (task #34). +// Lands once #34 clears. package strings; @@ -1611,11 +1617,11 @@ export fn trim(input: str, exclude: rune) str = { return ltrim(rtrim(input, exclude), exclude); }; -// iterator — forward UTF-8 rune cursor over a `str`. Layout flattens -// Hare's anonymous-embedded `utf8::decoder` -// (ref/hare/strings/iter.ha:6-9) to explicit fields; `reverse` is -// retained on the type because `riter` will populate it once `prev` / -// `utf8.prev` land. May be copied to save state. +// iterator — UTF-8 rune cursor over a `str`. Layout flattens Hare's +// anonymous-embedded `utf8::decoder` (ref/hare/strings/iter.ha:6-9) to +// explicit fields. `reverse` selects walk direction: forward iterators +// (`iter`) advance through utf8.next; reverse iterators (`riter`) advance +// through utf8.prev. May be copied to save state. export type iterator = struct { offs: i32, src: []u8, @@ -1632,25 +1638,82 @@ export fn iter(src: str) iterator = { return r; }; -// next — advance the iterator one rune. Returns `utf8.done` at end -// of input. Aborts on `more` / `invalid` — mirrors Hare's -// ref/hare/strings/iter.ha:51-58 `move()`, which aborts unconditionally -// on those arms ("Invalid UTF-8 string (this should not happen)"). -// -// Copy-in / copy-out is the cost of flattening the embedded decoder; -// see the iterator divergence note at the top of the file. -export fn next(it: *iterator) (rune | utf8.done) = { +// riter — initialize a reverse iterator at the end of `src`. `next` +// on a reverse iterator walks back through the string. +// ref/hare/strings/iter.ha:32. +export fn riter(src: str) iterator = { + let r: iterator; + r.src = toutf8(src); + r.offs = src.len; + r.reverse = true; + return r; +}; + +// move — private dispatch shared by next/prev. `forward` selects +// utf8.next vs utf8.prev. Aborts on more/invalid per Hare's +// ref/hare/strings/iter.ha:51-58 ("Invalid UTF-8 string (this should +// not happen)"). Hare picks the utf8 function via a fn-pointer; ww +// branches on `forward` at each call site instead. +fn move(forward: bool, it: *iterator) (rune | utf8.done) = { let d: utf8.decoder; d.src = it.src; d.offs = it.offs; - match (utf8.next(&d)) { - case let r: rune => { it.offs = d.offs; return r; }; - case let dn: utf8.done => return dn; - case let m: utf8.more => abort("strings.next: invalid UTF-8"); - case let e: utf8.invalid => abort("strings.next: invalid UTF-8"); + if (forward) { + match (utf8.next(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; + } else { + match (utf8.prev(&d)) { + case let r: rune => { it.offs = d.offs; return r; }; + case let dn: utf8.done => return dn; + case let m: utf8.more => abort("strings.move: invalid UTF-8"); + case let e: utf8.invalid => abort("strings.move: invalid UTF-8"); + }; }; }; +// next — advance the iterator one rune. Forward iterators step +// through utf8.next; reverse iterators (riter) step backward through +// utf8.prev. Returns utf8.done at end-of-walk. ref/hare/strings/iter.ha:45. +export fn next(it: *iterator) (rune | utf8.done) = { + return move(!it.reverse, it); +}; + +// prev — step back one rune. Dual to next: on a forward iterator +// this walks utf8.prev; on a reverse iterator (riter) it walks +// utf8.next. ref/hare/strings/iter.ha:49. +export fn prev(it: *iterator) (rune | utf8.done) = { + return move(it.reverse, it); +}; + +// iterstr — borrowed view of the bytes remaining in the iterator's +// walk direction. Forward iter: bytes from offs to end; reverse iter: +// bytes from start to offs. ref/hare/strings/iter.ha:63. +export fn iterstr(it: *iterator) str = { + let r: []u8; + if (it.reverse) { + r = it.src[0:it.offs]; + } else { + r = it.src[it.offs:it.src.len]; + }; + return fromutf8_unsafe(r); +}; + +// strings.slice deferred — Hare's delegation form +// `fromutf8_unsafe(utf8::slice(begin, end))` (ref/hare/strings/iter.ha:76) +// triggers wwstage fnretlookup same-name cross-module phantom return-ABI +// fixup (task #34, Class A wwstage UNDER, sub-bug of #4e). Inlining the +// body would diverge from Hare at API level (rule 9). Land after #34. + +// position — byte-wise offset of the iterator in its source. +// ref/hare/strings/iter.ha:82. +export fn position(it: *iterator) i32 = { + return it.offs; +}; + // strconv — number↔string conversions. // // Mirrors Hare's strconv:: surface. The *tos functions return a