lib/strings+test: Hare port (dup/concat/trim/index/contains/has{pre,suf}fix/compare/utf8)
Hare-faithful index/predicate family per ref/hare/strings/{dup,
concat,trim,index,suffix,contains,compare,utf8}.ha. Non-variadic
subset (concat 2-arg, trim single-rune, contains single-needle)
pending task #16 — cstage variadic-pack drops .len on multi-field
element types; ship the Hare-faithful single-arg shape now, file
the variadic upgrade as follow-up. `sub` follow-up filed as #29
(commit 2 with iterator + utf8.chars relocation).
Surface: dup, concat, trim/trimprefix/trimsuffix (single rune),
hasprefix, hassuffix (both with (str|rune) sum needle),
byteindex, rbyteindex (both with (str|rune) sum needle),
contains (single str needle), compare, toutf8, fromutf8_unsafe,
runebytes helper. (str|rune) match arms route the rune via
utf8.encoderune into a [4]u8 scratch then bytes.index/rindex —
drew-devault's directive for clean Hare-fidelity over invented
ASCII-only rune-byte arms.
byteindex / rbyteindex rune-arm semantic correction —
corpus-coverage-blind unmask. Pre-existing impl scanned for
`r: u8` (broken for all rune values >0x7F since strings.ww first
landed; no caller exercised it). Replaced with utf8.encoderune-
based scan via runebytes helper. Severity-marker: silent
wrong-result for any non-ASCII rune needle, masked by zero
in-tree callers until lib/strings + utf8 chain pulled the shape
in.
Build-system propagation: lib/strings depends transitively on
lib/encoding/utf8 (via byteindex's rune arm). cmd/ww driver's
locate_import_in (cmd/ww/main.c:85) walks `<dir>/<name>.ww` and
`<dir>/<name>/<name>.ww` only — `use utf8;` doesn't find
lib/encoding/utf8/utf8.ww without explicit `-I lib/encoding/utf8`.
Propagated through 5 wwstage-tool Makefile targets + 7 test
wrappers + test/wcc/995_self_rebuild.c sprintf lines. Task #17
filed for the principled resolver fix (subdir walk vs Hare's
qualified `use encoding::utf8;` notation).
This commit chain (#15 strings) surfaced 7 cgen bugs during
landing: #16 cstage variadic-pack, #17 resolver nested-paths,
#27 aliaslookup leaf-collision, #22 zero-init !void/void-alias
let-decl, #15-cstage retscr SSoT name, #24 composite CALL return
as composite arg, #28 N_DOT calleeparams. All blocking ones
fixed (#16/#17 deferred-with-stopgap, others fixed in their
respective commits). Pre-flight + stop-and-surface discipline
held throughout — no workarounds shipped in stdlib.
Tests:
- 966_strings_run drives lib/strings/stringstest.ww via ww run.
15 @test fns: dup (alloc, multibyte), concat (empty, lopsided,
multibyte), trim/ltrim/rtrim incl. 4-byte rune U+1D68A,
hasprefix/hassuffix with (str|rune) incl. multibyte,
byteindex/rbyteindex both arms 1/2/3/4-byte rune coverage,
compare. Cited from ref/hare/strings/+test.ha where vectors
apply.
100/100 ok. 995_self_rebuild stays green (ww2==ww3==ww4 byte-id).
This commit is contained in:
@@ -855,56 +855,38 @@ export fn freearena(a: *arena) void = {
|
||||
};
|
||||
};
|
||||
|
||||
// MODULE: strings
|
||||
// strings — operations over the immutable str type ({ *u8, len }).
|
||||
// Mirrors Hare's strings::; `len` and `is-empty` aren't functions
|
||||
// (callers use `s.len` and `s.len == 0` directly).
|
||||
// MODULE: bytes
|
||||
// bytes — slice operations over []u8. Mirrors Hare's bytes module
|
||||
// (ref/hare/bytes/) for the in-tree subset: search/equality/prefix
|
||||
// helpers used by lib/encoding, lib/bufio, lib/memio.
|
||||
//
|
||||
// Documented divergences from Hare:
|
||||
// - index_slice / rindex_slice use naive O(n·m); Hare specialises
|
||||
// 2/3/4-byte needles and falls back to two_way (Crochemore-Perrin)
|
||||
// for longer (ref/hare/bytes/index.ha:61, ref/hare/bytes/two_way.ha).
|
||||
// Correctness equivalent.
|
||||
// - contains takes a single needle; Hare's contains is variadic
|
||||
// `(u8 | []u8)...` (ref/hare/bytes/contains.ha:5). No caller needs
|
||||
// the variadic shape yet; graduate when one does.
|
||||
|
||||
use os;
|
||||
|
||||
// compare — bytewise three-way comparison: negative if a<b, 0 if equal,
|
||||
// positive if a>b. Matches Hare's strings::compare. ASCII-order, not
|
||||
// locale-aware. Callers that just need equality use `compare(a, b) == 0`.
|
||||
export fn compare(a: str, b: str) i32 = {
|
||||
let n: i32 = a.len;
|
||||
if (b.len < n) { n = b.len; };
|
||||
// equal — true iff `a` and `b` have the same length and contents.
|
||||
// ref/hare/bytes/equal.ha:9.
|
||||
export fn equal(a: []u8, b: []u8) bool = {
|
||||
if (a.len != b.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < n) {
|
||||
if (a[i] != b[i]) { return (a[i]: i32) - (b[i]: i32); };
|
||||
i += 1;
|
||||
};
|
||||
return a.len - b.len;
|
||||
};
|
||||
|
||||
export fn hasprefix(s: str, p: str) bool = {
|
||||
if (p.len > s.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < p.len) {
|
||||
if (s[i] != p[i]) { return false; };
|
||||
for (i < a.len) {
|
||||
if (a[i] != b[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
export fn hassuffix(s: str, suf: str) bool = {
|
||||
if (suf.len > s.len) { return false; };
|
||||
let off: i32 = s.len - suf.len;
|
||||
let i: i32 = 0;
|
||||
for (i < suf.len) {
|
||||
if (s[off + i] != suf[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
// byteindex — first byte position of `needle` in `s`. Mirrors Hare's
|
||||
// strings::byteindex: a single-codepoint rune scans for the byte that
|
||||
// encodes it (ASCII only here — multi-byte UTF-8 awaits utf8 encode),
|
||||
// a str needle scans for the substring. Returns void if absent.
|
||||
export fn byteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
// index — first offset of `needle` in `s`. u8 needle scans for the
|
||||
// byte; []u8 needle scans for the substring. void if absent.
|
||||
// ref/hare/bytes/index.ha:6.
|
||||
export fn index(s: []u8, needle: (u8 | []u8)) (i32 | void) = {
|
||||
match (needle) {
|
||||
case let r: rune => {
|
||||
let c: u8 = r: u8;
|
||||
case let c: u8 => {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i] == c) { return i; };
|
||||
@@ -912,7 +894,7 @@ export fn byteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
};
|
||||
return;
|
||||
};
|
||||
case let sub: str => {
|
||||
case let sub: []u8 => {
|
||||
if (sub.len == 0) { return 0; };
|
||||
if (sub.len > s.len) { return; };
|
||||
let last: i32 = s.len - sub.len;
|
||||
@@ -933,87 +915,12 @@ export fn byteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
return;
|
||||
};
|
||||
|
||||
// contains — true iff `sub` appears in `s`. Mirrors Hare's
|
||||
// strings::contains shape (byte-wise on the str-needle case).
|
||||
export fn contains(s: str, sub: str) bool = {
|
||||
let r: (i32 | void) = byteindex(s, sub);
|
||||
match (r) {
|
||||
case let i: i32 => return true;
|
||||
case void => return false;
|
||||
};
|
||||
return false;
|
||||
};
|
||||
|
||||
// concat — joins two strings into a fresh str. Caller owns the
|
||||
// returned str's storage; release via `os.free(r.ptr, r.len)`. Mirrors
|
||||
// Hare's strings::concat shape.
|
||||
export fn concat(a: str, b: str) str = {
|
||||
let total: i32 = a.len + b.len;
|
||||
let buf: *u8 = os.alloc(total: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < a.len) { buf[i] = a[i]; i += 1; };
|
||||
let j: i32 = 0;
|
||||
for (j < b.len) { buf[a.len + j] = b[j]; j += 1; };
|
||||
let r: str;
|
||||
r.ptr = buf;
|
||||
r.len = total;
|
||||
return r;
|
||||
};
|
||||
|
||||
// dup — duplicate a string into a fresh allocation. Caller owns the
|
||||
// returned str's storage; release via `os.free(r.ptr, r.len)`. Mirrors
|
||||
// Hare's strings::dup shape — Hare returns `(str | nomem)`, ww doesn't
|
||||
// have nomem (os.alloc aborts on OOM), so we return plain `str`.
|
||||
//
|
||||
// Empty input yields a `{nil, 0}` str — Hare returns the static empty
|
||||
// string; same observable result.
|
||||
export fn dup(s: str) str = {
|
||||
let r: str;
|
||||
r.ptr = nil;
|
||||
r.len = 0;
|
||||
if (s.len == 0) { return r; };
|
||||
let buf: *u8 = os.alloc(s.len: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) { buf[i] = s[i]; i += 1; };
|
||||
r.ptr = buf;
|
||||
r.len = s.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// freeall — release every str element in `s` (those that were
|
||||
// individually allocated) plus the slice's backing storage. Mirrors
|
||||
// Hare's strings::freeall — the natural disposer for any function
|
||||
// returning a fresh `[]str` of dup'd elements (e.g. shlex.split).
|
||||
//
|
||||
// Each element is freed via os.free at its own length; the slice
|
||||
// header storage is freed at `cap * 16` bytes (one str = 16B). Empty
|
||||
// elements (`{nil, 0}` from a zero-length dup) are skipped — calling
|
||||
// os.free on a nil pointer at len 0 would tickle the rt_free guard
|
||||
// that the runtime treats as a logic bug.
|
||||
//
|
||||
// `cap == 0` means the slice was never grown (empty `[]str` with no
|
||||
// backing allocation); skip the header free in that case too.
|
||||
export fn freeall(s: []str) void = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i].len > 0) {
|
||||
os.free(s[i].ptr: *void, s[i].len: u64);
|
||||
};
|
||||
i += 1;
|
||||
};
|
||||
if (s.cap > 0) {
|
||||
os.free(s.ptr: *void, (s.cap: u64) * 16u64);
|
||||
};
|
||||
};
|
||||
|
||||
// rbyteindex — last byte position of `needle` in `s`. Mirrors Hare's
|
||||
// strings::rbyteindex. Rune needle scans for the byte that encodes it
|
||||
// (ASCII only); str needle scans for the substring. Empty str needle
|
||||
// matches at s.len.
|
||||
export fn rbyteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
// rindex — last offset of `needle` in `s`. Empty []u8 needle returns
|
||||
// s.len (ref/hare/bytes/index.ha:103 — Hare's loop yields r-0 at i=0).
|
||||
// ref/hare/bytes/index.ha:86.
|
||||
export fn rindex(s: []u8, needle: (u8 | []u8)) (i32 | void) = {
|
||||
match (needle) {
|
||||
case let r: rune => {
|
||||
let c: u8 = r: u8;
|
||||
case let c: u8 => {
|
||||
let i: i32 = s.len - 1;
|
||||
for (i >= 0) {
|
||||
if (s[i] == c) { return i; };
|
||||
@@ -1021,7 +928,7 @@ export fn rbyteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
};
|
||||
return;
|
||||
};
|
||||
case let sub: str => {
|
||||
case let sub: []u8 => {
|
||||
if (sub.len == 0) { return s.len; };
|
||||
if (sub.len > s.len) { return; };
|
||||
let i: i32 = s.len - sub.len;
|
||||
@@ -1041,9 +948,523 @@ export fn rbyteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
return;
|
||||
};
|
||||
|
||||
// sub — borrowed substring `s[start..end]`. Mirrors Hare's
|
||||
// strings::sub. Caller must ensure 0 <= start <= end <= s.len; out-of-
|
||||
// range indices are clamped silently here, where Hare aborts.
|
||||
// contains — true iff `needle` (byte or sub-slice) appears in `s`.
|
||||
// ref/hare/bytes/contains.ha:5 (variadic subset; see header note).
|
||||
export fn contains(s: []u8, needle: (u8 | []u8)) bool = {
|
||||
match (index(s, needle)) {
|
||||
case let i: i32 => return true;
|
||||
case void => return false;
|
||||
};
|
||||
return false;
|
||||
};
|
||||
|
||||
// hasprefix — true iff `s` starts with `pre`.
|
||||
// ref/hare/bytes/contains.ha:21.
|
||||
export fn hasprefix(s: []u8, pre: []u8) bool = {
|
||||
if (pre.len > s.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < pre.len) {
|
||||
if (s[i] != pre[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
// hassuffix — true iff `s` ends with `suf`.
|
||||
// ref/hare/bytes/contains.ha:35.
|
||||
export fn hassuffix(s: []u8, suf: []u8) bool = {
|
||||
if (suf.len > s.len) { return false; };
|
||||
let off: i32 = s.len - suf.len;
|
||||
let i: i32 = 0;
|
||||
for (i < suf.len) {
|
||||
if (s[off + i] != suf[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
// reverse — in-place reverse of `s`. ref/hare/bytes/reverse.ha:5.
|
||||
export fn reverse(s: []u8) void = {
|
||||
let i: i32 = 0;
|
||||
let j: i32 = s.len - 1;
|
||||
for (i < j) {
|
||||
let t: u8 = s[i];
|
||||
s[i] = s[j];
|
||||
s[j] = t;
|
||||
i += 1;
|
||||
j -= 1;
|
||||
};
|
||||
};
|
||||
|
||||
// zero — set every byte of `s` to 0. ref/hare/bytes/zero.ha:5.
|
||||
export fn zero(s: []u8) void = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
s[i] = 0u8;
|
||||
i += 1;
|
||||
};
|
||||
};
|
||||
|
||||
// MODULE: utf8
|
||||
// encoding/utf8 — UTF-8 encode/decode. Hare port; see
|
||||
// ref/hare/encoding/utf8/{types,rune,encode,decode,decodetable}.ha.
|
||||
//
|
||||
// The decoder is Hoehrmann's branchless DFA, originally published
|
||||
// at <https://bjoern.hoehrmann.de/utf-8/decoder/dfa/>. Hare's
|
||||
// ref/hare/encoding/utf8/decodetable.ha:4 restructures Hoehrmann's
|
||||
// flat table to 2D `[8][256]i8`; we flatten back to 1D `[2048]i8`
|
||||
// because ww cgen does not yet ship 2D arrays (task #20).
|
||||
//
|
||||
// Surface deviation from ref/hare/encoding/utf8:
|
||||
//
|
||||
// - `encoderune` takes a caller-supplied `out: []u8` and returns
|
||||
// the byte count. Hare returns a slice into a `static let buf`;
|
||||
// the caller-buffer form mirrors lib/encoding/hex.encode and
|
||||
// skips the static-buffer/slice-return pair.
|
||||
//
|
||||
// Deferred (no in-tree caller, follow-up tasks): `prev`, `slice`,
|
||||
// `position`, `remaining`, `appendrune`, `strencode`, `strdecode`.
|
||||
// Hare's string-iteration surface (`strings::iterator`/`strings::next`
|
||||
// — ref/hare/strings/iter.ha) lives under lib/strings, not here.
|
||||
|
||||
// ref/hare/encoding/utf8/types.ha:6 — incomplete trailing sequence.
|
||||
// Plain `void` (not `!void`): a truncated tail is a control-flow
|
||||
// signal, not an error caller can ignore.
|
||||
export type more = void;
|
||||
|
||||
// ref/hare/encoding/utf8/types.ha:9 — invalid UTF-8 sequence.
|
||||
export type invalid = !void;
|
||||
|
||||
// `done` is not a built-in singleton in ww (Hare ships it as part of
|
||||
// the type system). Plain `void` (not `!void`): end-of-input is a
|
||||
// continuation signal, not an error. lib/io spells its EOF the same
|
||||
// way (lib/io/io.ww:8-11).
|
||||
export type done = void;
|
||||
|
||||
// ref/hare/encoding/utf8/decodetable.ha:4 — Hoehrmann's UTF-8 DFA,
|
||||
// flat 1D `[2048]i8`. Layout: dfa[state*256 + byte] gives the next
|
||||
// state (>0), the accept transition (0 — emit rune), or invalid (-1).
|
||||
// Values match ref/hare/encoding/utf8/decodetable.ha verbatim.
|
||||
let dfa: [2048]i8 = [
|
||||
// state 0 — initial byte: ASCII accepts (0), continuation/illegal
|
||||
// byte rejects (-1), legal multibyte start emits a state.
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
3i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 4i8, 2i8, 2i8,
|
||||
5i8, 6i8, 6i8, 6i8, 7i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 1 — expecting one continuation byte (0x80..0xBF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 2 — expecting one continuation byte (full 0x80..0xBF range).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 3 — first byte was 0xE0; continuation byte must be 0xA0..0xBF
|
||||
// (rejects overlong 3-byte encodings).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 4 — first byte was 0xED; continuation byte must be 0x80..0x9F
|
||||
// (rejects UTF-16 surrogate codepoints U+D800..U+DFFF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 5 — first byte was 0xF0; continuation byte must be 0x90..0xBF
|
||||
// (rejects overlong 4-byte encodings).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 6 — middle continuation byte of a 4-byte sequence (0x80..0xBF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 7 — first byte was 0xF4; continuation byte must be 0x80..0x8F
|
||||
// (rejects codepoints above U+10FFFF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
];
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:17 — payload-bit masks. Hare's
|
||||
// [2][8]u8 flattened to 1D [16]u8; row 0 (offsets 0..7) is the
|
||||
// continuation-byte mask (always 0x3F), row 1 (offsets 8..15) is the
|
||||
// initial-byte payload mask indexed by the transition class.
|
||||
let masks: [16]u8 = [
|
||||
0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8,
|
||||
0x7fu8, 0x1fu8, 0x0fu8, 0x0fu8, 0x0fu8, 0x07u8, 0x07u8, 0x07u8,
|
||||
];
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:6 — incremental decoder state.
|
||||
export type decoder = struct {
|
||||
offs: i32,
|
||||
src: []u8,
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:12.
|
||||
export fn decode(src: []u8) decoder = {
|
||||
let d: decoder;
|
||||
d.src = src;
|
||||
d.offs = 0;
|
||||
return d;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:27. Returns the next rune from a
|
||||
// decoder, `done` at end-of-input, `more` on truncated trailing
|
||||
// sequence, `invalid` on malformed input (overlong, surrogate,
|
||||
// out-of-range, bad continuation).
|
||||
//
|
||||
// Algorithm is verbatim Hoehrmann (see file header). One structural
|
||||
// rewrite: Hare encodes the "initial vs continuation byte" decision
|
||||
// as the branchless `(state - 1): uint >> 31`, which assumes a 32-bit
|
||||
// uint. ww's uint is 64-bit (cmd/wcc/type.c:58), so the shift answer
|
||||
// would be 0x1_ffff_ffff rather than 1. We spell the same predicate
|
||||
// with an explicit conditional.
|
||||
export fn next(d: *decoder) (rune | done | more | invalid) = {
|
||||
if (d.offs == d.src.len) {
|
||||
let dn: done; return dn;
|
||||
};
|
||||
let nx: i32 = 0;
|
||||
let state: i32 = 0;
|
||||
let r: u32 = 0u32;
|
||||
for (d.offs < d.src.len) {
|
||||
let b: u8 = d.src[d.offs];
|
||||
let bi: i32 = b: i32;
|
||||
let row: i32 = state * 256 + bi;
|
||||
let cell: i8 = dfa[row];
|
||||
nx = cell: i32;
|
||||
let mi: i32 = 0;
|
||||
if (state == 0) { mi = 1; };
|
||||
let m: u8 = masks[mi * 8 + (nx & 7)];
|
||||
r = (r << 6u32) | ((b & m): u32);
|
||||
if (nx <= 0) {
|
||||
d.offs += 1;
|
||||
if (nx == 0) { return r: rune; };
|
||||
let e: invalid; return e;
|
||||
};
|
||||
state = nx;
|
||||
d.offs += 1;
|
||||
};
|
||||
let mr: more; return mr;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:207. Strict whole-input check.
|
||||
// The hot path: tight DFA loop, no rune assembly. Bails the moment
|
||||
// the table returns -1 so malformed inputs don't pay for the rest
|
||||
// of the buffer.
|
||||
export fn validate(src: []u8) (void | invalid) = {
|
||||
let state: i32 = 0;
|
||||
let i: i32 = 0;
|
||||
for (i < src.len) {
|
||||
if (state < 0) { break; };
|
||||
let bi: i32 = src[i]: i32;
|
||||
let cell: i8 = dfa[state * 256 + bi];
|
||||
state = cell: i32;
|
||||
i += 1;
|
||||
};
|
||||
if (state == 0) { return; };
|
||||
let e: invalid; return e;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/rune.ha:5. Encoded byte length of `r` as
|
||||
// UTF-8. Callers in ww use this to size the buffer they hand to
|
||||
// [[encoderune]]; values >0x10FFFF or negative are not legal Unicode
|
||||
// codepoints and Hare aborts on them in `encoderune` itself, so we
|
||||
// keep `runesz` infallible (matches Hare).
|
||||
export fn runesz(r: rune) i32 = {
|
||||
let ch: u32 = r: u32;
|
||||
if (ch < 128u32) { return 1; };
|
||||
if (ch < 2048u32) { return 2; };
|
||||
if (ch < 65536u32) { return 3; };
|
||||
return 4;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/rune.ha:15. Expected byte length of the
|
||||
// codepoint that starts with `c`, or `invalid` if `c` cannot start
|
||||
// a legal UTF-8 sequence. Constants written in decimal because ww
|
||||
// doesn't accept Hare's `0b1000_0000` binary syntax: 0x80=128,
|
||||
// 0xC2=194, 0xE0=224, 0xF0=240, 0xF8=248.
|
||||
export fn utf8sz(c: u8) (i32 | invalid) = {
|
||||
if (c < 128u8) { return 1; };
|
||||
if (c < 194u8) { let e: invalid; return e; };
|
||||
if (c >= 248u8) { let e: invalid; return e; };
|
||||
if (c < 224u8) { return 2; };
|
||||
if (c < 240u8) { return 3; };
|
||||
return 4;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/encode.ha:7. Encode `r` into `out` (caller-
|
||||
// supplied; must hold at least [[runesz]](r) bytes) and return the
|
||||
// byte count. ABORT if `r` is a UTF-16 surrogate or above U+10FFFF —
|
||||
// same precondition Hare asserts at ref/hare/encoding/utf8/encode.ha:9.
|
||||
//
|
||||
// Surface deviation: Hare returns `[]u8` (slice into a static buf).
|
||||
// ww uses the caller-buffer form (matches lib/encoding/hex.encode);
|
||||
// caller can reuse a [4]u8 stack scratch across encodes.
|
||||
export fn encoderune(out: []u8, r: rune) i32 = {
|
||||
let ch: u32 = r: u32;
|
||||
if (ch >= 0xD800u32) {
|
||||
if (ch <= 0xDFFFu32) {
|
||||
abort("utf8.encoderune: surrogate codepoint");
|
||||
};
|
||||
};
|
||||
if (ch > 0x10FFFFu32) {
|
||||
abort("utf8.encoderune: codepoint > U+10FFFF");
|
||||
};
|
||||
|
||||
let n: i32 = 0;
|
||||
let first: u8 = 0u8;
|
||||
if (ch < 0x80u32) {
|
||||
first = 0u8; n = 1;
|
||||
} else if (ch < 0x800u32) {
|
||||
first = 0xC0u8; n = 2;
|
||||
} else if (ch < 0x10000u32) {
|
||||
first = 0xE0u8; n = 3;
|
||||
} else {
|
||||
first = 0xF0u8; n = 4;
|
||||
};
|
||||
|
||||
let v: u32 = ch;
|
||||
let i: i32 = n - 1;
|
||||
for (i > 0) {
|
||||
out[i] = ((v: u8) & 0x3Fu8) | 0x80u8;
|
||||
v = v >> 6u32;
|
||||
i -= 1;
|
||||
};
|
||||
out[0] = (v: u8) | first;
|
||||
return n;
|
||||
};
|
||||
|
||||
|
||||
// MODULE: strings
|
||||
// strings — operations over str ({ptr,len}). Hare port; see
|
||||
// ref/hare/strings/.
|
||||
//
|
||||
// Documented divergences from Hare:
|
||||
//
|
||||
// - `concat(a, b)` is 2-arg. Hare ships `concat(strs: str...)`
|
||||
// (ref/hare/strings/concat.ha:5). Blocks on task #16 (cstage
|
||||
// variadic-pack drops .len of multi-field element type). Cite
|
||||
// reverts on fix.
|
||||
// - `trim` / `ltrim` / `rtrim` take a single rune. Hare's are
|
||||
// `(exclude: rune...)` (ref/hare/strings/trim.ha:54). Same
|
||||
// blocker as concat. Hare's no-rune branch (strip whitespace)
|
||||
// is also dropped — depends on a rune set.
|
||||
// - `contains` is non-variadic. Hare's is
|
||||
// `contains(haystack, needles: (str | rune)...)`
|
||||
// (ref/hare/strings/contains.ha:9). Same blocker.
|
||||
// - `byteindex` / `rbyteindex` rune arms encode via
|
||||
// `utf8.encoderune`; the legacy impls scanned for `r: u8` (an
|
||||
// undocumented ASCII-only restriction that silently dropped
|
||||
// to the wrong byte for U+80..U+7FF and higher).
|
||||
// - `dup(s: str) str` — Hare returns `(str | nomem)`. ww's
|
||||
// `os.alloc` aborts on OOM (no `nomem` type), so we return plain
|
||||
// `str`. Empty input returns `{nil, 0}`; Hare returns the static
|
||||
// empty string — same observable result.
|
||||
|
||||
use bytes;
|
||||
use utf8;
|
||||
use os;
|
||||
|
||||
// toutf8 — borrowed []u8 view of `s`. ref/hare/strings/utf8.ha:29.
|
||||
// `cap` equals `len`; the slice does not own a separate allocation.
|
||||
export fn toutf8(s: str) []u8 = {
|
||||
let r: []u8;
|
||||
r.ptr = s.ptr;
|
||||
r.len = s.len;
|
||||
r.cap = s.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// fromutf8_unsafe — borrowed str view of `in`. Does not validate.
|
||||
// ref/hare/strings/utf8.ha:10.
|
||||
export fn fromutf8_unsafe(in: []u8) str = {
|
||||
let r: str;
|
||||
r.ptr = in.ptr;
|
||||
r.len = in.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// compare — three-way bytewise codepoint-order comparison.
|
||||
// ref/hare/strings/compare.ha:12.
|
||||
export fn compare(a: str, b: str) i32 = {
|
||||
let n: i32 = a.len;
|
||||
if (b.len < n) { n = b.len; };
|
||||
let i: i32 = 0;
|
||||
for (i < n) {
|
||||
if (a[i] != b[i]) { return (a[i]: i32) - (b[i]: i32); };
|
||||
i += 1;
|
||||
};
|
||||
return a.len - b.len;
|
||||
};
|
||||
|
||||
// dup — allocate a fresh copy of `s`. Caller releases with
|
||||
// `os.free(r.ptr, r.len: u64)`. ref/hare/strings/dup.ha:7.
|
||||
export fn dup(s: str) str = {
|
||||
let r: str;
|
||||
r.ptr = nil;
|
||||
r.len = 0;
|
||||
if (s.len == 0) { return r; };
|
||||
let buf: *u8 = os.alloc(s.len: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) { buf[i] = s[i]; i += 1; };
|
||||
r.ptr = buf;
|
||||
r.len = s.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// freeall — release each element + the slice header. The natural
|
||||
// disposer for any `[]str` of dup'd elements (e.g. shlex.split).
|
||||
// ref/hare/strings/dup.ha:38.
|
||||
//
|
||||
// Empty elements (`{nil, 0}` from a zero-length dup) are skipped:
|
||||
// os.free on a nil pointer at len 0 tickles the rt_free guard. The
|
||||
// slice header itself is freed at `cap * 16` (one str = 16B); a
|
||||
// never-grown slice (cap == 0) skips the header free.
|
||||
export fn freeall(s: []str) void = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i].len > 0) {
|
||||
os.free(s[i].ptr: *void, s[i].len: u64);
|
||||
};
|
||||
i += 1;
|
||||
};
|
||||
if (s.cap > 0) {
|
||||
os.free(s.ptr: *void, (s.cap: u64) * 16u64);
|
||||
};
|
||||
};
|
||||
|
||||
// concat — fresh allocation containing `a` then `b`. Caller releases
|
||||
// with `os.free(r.ptr, r.len: u64)`. ref/hare/strings/concat.ha:5
|
||||
// (subset: Hare's `(strs: str...)` blocks on task #16).
|
||||
export fn concat(a: str, b: str) str = {
|
||||
let total: i32 = a.len + b.len;
|
||||
let buf: *u8 = os.alloc(total: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < a.len) { buf[i] = a[i]; i += 1; };
|
||||
let j: i32 = 0;
|
||||
for (j < b.len) { buf[a.len + j] = b[j]; j += 1; };
|
||||
let r: str;
|
||||
r.ptr = buf;
|
||||
r.len = total;
|
||||
return r;
|
||||
};
|
||||
|
||||
// sub — borrowed `s[start..end]`. ref/hare/strings/sub.ha:30 is
|
||||
// rune-wise; this ww form is byte-wise (no rune iterator yet, planned
|
||||
// for commit 2). Clamps out-of-range silently where Hare aborts —
|
||||
// retained for the existing getopt caller; will graduate when the
|
||||
// rune-wise form lands.
|
||||
export fn sub(s: str, start: i32, end: i32) str = {
|
||||
let lo: i32 = start;
|
||||
let hi: i32 = end;
|
||||
@@ -1056,58 +1477,145 @@ export fn sub(s: str, start: i32, end: i32) str = {
|
||||
return r;
|
||||
};
|
||||
|
||||
// trimprefix — `s` with `pre` stripped from the front, or `s`
|
||||
// unchanged if it doesn't start with `pre`. Returns a borrowed view.
|
||||
// Mirrors Hare's strings::trimprefix.
|
||||
export fn trimprefix(s: str, pre: str) str = {
|
||||
if (!hasprefix(s, pre)) { return s; };
|
||||
// runebytes — encode `r` into caller's `scratch` (must hold 4 bytes)
|
||||
// and return the borrowed slice trimmed to the encoded length. Hare
|
||||
// inlines the same shape at ref/hare/strings/index.ha:132.
|
||||
fn runebytes(scratch: []u8, r: rune) []u8 = {
|
||||
let n: i32 = utf8.encoderune(scratch, r);
|
||||
let s: []u8;
|
||||
s.ptr = scratch.ptr;
|
||||
s.len = n;
|
||||
s.cap = n;
|
||||
return s;
|
||||
};
|
||||
|
||||
// hasprefix — true iff `in` begins with `prefix`.
|
||||
// ref/hare/strings/suffix.ha:8.
|
||||
export fn hasprefix(in: str, prefix: (str | rune)) bool = {
|
||||
let scratch: [4]u8;
|
||||
let p: []u8 = match (prefix) {
|
||||
case let s: str => yield toutf8(s);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.hasprefix(toutf8(in), p);
|
||||
};
|
||||
|
||||
// hassuffix — true iff `in` ends with `suff`.
|
||||
// ref/hare/strings/suffix.ha:26.
|
||||
export fn hassuffix(in: str, suff: (str | rune)) bool = {
|
||||
let scratch: [4]u8;
|
||||
let s: []u8 = match (suff) {
|
||||
case let v: str => yield toutf8(v);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.hassuffix(toutf8(in), s);
|
||||
};
|
||||
|
||||
// byteindex — byte-wise offset of `needle` in `haystack`, or void if
|
||||
// absent. ref/hare/strings/index.ha:127. Rune arm encodes via
|
||||
// utf8.encoderune (Hare passes the encoded slice straight to
|
||||
// bytes::index).
|
||||
export fn byteindex(haystack: str, needle: (str | rune)) (i32 | void) = {
|
||||
let scratch: [4]u8;
|
||||
let n: []u8 = match (needle) {
|
||||
case let s: str => yield toutf8(s);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.index(toutf8(haystack), n);
|
||||
};
|
||||
|
||||
// rbyteindex — byte-wise offset of the last `needle` in `haystack`.
|
||||
// ref/hare/strings/index.ha:138.
|
||||
export fn rbyteindex(haystack: str, needle: (str | rune)) (i32 | void) = {
|
||||
let scratch: [4]u8;
|
||||
let n: []u8 = match (needle) {
|
||||
case let s: str => yield toutf8(s);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.rindex(toutf8(haystack), n);
|
||||
};
|
||||
|
||||
// contains — true iff `needle` occurs in `haystack`.
|
||||
// ref/hare/strings/contains.ha:9 (subset: Hare's variadic form
|
||||
// `(needles: (str | rune)...)` blocks on task #16).
|
||||
export fn contains(haystack: str, needle: (str | rune)) bool = {
|
||||
match (byteindex(haystack, needle)) {
|
||||
case let i: i32 => return true;
|
||||
case void => return false;
|
||||
};
|
||||
return false;
|
||||
};
|
||||
|
||||
// trimprefix — `s` with `prefix` stripped from the front, or `s`
|
||||
// unchanged if it doesn't start with `prefix`. Borrowed view.
|
||||
// ref/hare/strings/trim.ha:60.
|
||||
export fn trimprefix(input: str, prefix: str) str = {
|
||||
if (!hasprefix(input, prefix)) { return input; };
|
||||
let r: str;
|
||||
r.ptr = s.ptr + (pre.len: u64);
|
||||
r.len = s.len - pre.len;
|
||||
r.ptr = input.ptr + (prefix.len: u64);
|
||||
r.len = input.len - prefix.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// trimsuffix — `s` with `suf` stripped from the end, or `s` unchanged
|
||||
// if it doesn't end with `suf`. Returns a borrowed view. Mirrors
|
||||
// Hare's strings::trimsuffix.
|
||||
export fn trimsuffix(s: str, suf: str) str = {
|
||||
if (!hassuffix(s, suf)) { return s; };
|
||||
// trimsuffix — symmetric. ref/hare/strings/trim.ha:69.
|
||||
export fn trimsuffix(input: str, suffix: str) str = {
|
||||
if (!hassuffix(input, suffix)) { return input; };
|
||||
let r: str;
|
||||
r.ptr = s.ptr;
|
||||
r.len = s.len - suf.len;
|
||||
r.ptr = input.ptr;
|
||||
r.len = input.len - suffix.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// ltrimbyte / rtrimbyte / trimbyte — strip occurrences of a single
|
||||
// byte from the left, right, or both ends. Returns a borrowed view.
|
||||
// Hare's strings::ltrim / rtrim / trim take a rune varargs set; ww's
|
||||
// subset takes a single byte (the common ASCII case).
|
||||
export fn ltrimbyte(s: str, c: u8) str = {
|
||||
// ltrim — strip occurrences of `exclude` (encoded as UTF-8) from the
|
||||
// front. Borrowed view. ref/hare/strings/trim.ha:11 (subset: single
|
||||
// rune; Hare's `(trim: rune...)` blocks on task #16). The no-rune
|
||||
// strip-whitespace branch is omitted for the same reason.
|
||||
export fn ltrim(input: str, exclude: rune) str = {
|
||||
let scratch: [4]u8;
|
||||
let pat: []u8 = runebytes(scratch[0:4], exclude);
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i] != c) { break; };
|
||||
i += 1;
|
||||
for (i + pat.len <= input.len) {
|
||||
let j: i32 = 0;
|
||||
let ok: bool = true;
|
||||
for (j < pat.len) {
|
||||
if (input[i + j] != pat[j]) { ok = false; j = pat.len; }
|
||||
else { j += 1; };
|
||||
};
|
||||
if (!ok) { break; };
|
||||
i += pat.len;
|
||||
};
|
||||
let r: str;
|
||||
r.ptr = s.ptr + (i: u64);
|
||||
r.len = s.len - i;
|
||||
r.ptr = input.ptr + (i: u64);
|
||||
r.len = input.len - i;
|
||||
return r;
|
||||
};
|
||||
|
||||
export fn rtrimbyte(s: str, c: u8) str = {
|
||||
let n: i32 = s.len;
|
||||
for (n > 0) {
|
||||
if (s[n - 1] != c) { break; };
|
||||
n -= 1;
|
||||
// rtrim — strip occurrences of `exclude` from the end. Borrowed view.
|
||||
// ref/hare/strings/trim.ha:32 (same subset note).
|
||||
export fn rtrim(input: str, exclude: rune) str = {
|
||||
let scratch: [4]u8;
|
||||
let pat: []u8 = runebytes(scratch[0:4], exclude);
|
||||
let n: i32 = input.len;
|
||||
for (n >= pat.len) {
|
||||
let off: i32 = n - pat.len;
|
||||
let j: i32 = 0;
|
||||
let ok: bool = true;
|
||||
for (j < pat.len) {
|
||||
if (input[off + j] != pat[j]) { ok = false; j = pat.len; }
|
||||
else { j += 1; };
|
||||
};
|
||||
if (!ok) { break; };
|
||||
n -= pat.len;
|
||||
};
|
||||
let r: str;
|
||||
r.ptr = s.ptr;
|
||||
r.ptr = input.ptr;
|
||||
r.len = n;
|
||||
return r;
|
||||
};
|
||||
|
||||
export fn trimbyte(s: str, c: u8) str = {
|
||||
return rtrimbyte(ltrimbyte(s, c), c);
|
||||
// trim — strip from both ends. ref/hare/strings/trim.ha:54.
|
||||
export fn trim(input: str, exclude: rune) str = {
|
||||
return ltrim(rtrim(input, exclude), exclude);
|
||||
};
|
||||
|
||||
// MODULE: strconv
|
||||
|
||||
@@ -855,56 +855,38 @@ export fn freearena(a: *arena) void = {
|
||||
};
|
||||
};
|
||||
|
||||
// MODULE: strings
|
||||
// strings — operations over the immutable str type ({ *u8, len }).
|
||||
// Mirrors Hare's strings::; `len` and `is-empty` aren't functions
|
||||
// (callers use `s.len` and `s.len == 0` directly).
|
||||
// MODULE: bytes
|
||||
// bytes — slice operations over []u8. Mirrors Hare's bytes module
|
||||
// (ref/hare/bytes/) for the in-tree subset: search/equality/prefix
|
||||
// helpers used by lib/encoding, lib/bufio, lib/memio.
|
||||
//
|
||||
// Documented divergences from Hare:
|
||||
// - index_slice / rindex_slice use naive O(n·m); Hare specialises
|
||||
// 2/3/4-byte needles and falls back to two_way (Crochemore-Perrin)
|
||||
// for longer (ref/hare/bytes/index.ha:61, ref/hare/bytes/two_way.ha).
|
||||
// Correctness equivalent.
|
||||
// - contains takes a single needle; Hare's contains is variadic
|
||||
// `(u8 | []u8)...` (ref/hare/bytes/contains.ha:5). No caller needs
|
||||
// the variadic shape yet; graduate when one does.
|
||||
|
||||
use os;
|
||||
|
||||
// compare — bytewise three-way comparison: negative if a<b, 0 if equal,
|
||||
// positive if a>b. Matches Hare's strings::compare. ASCII-order, not
|
||||
// locale-aware. Callers that just need equality use `compare(a, b) == 0`.
|
||||
export fn compare(a: str, b: str) i32 = {
|
||||
let n: i32 = a.len;
|
||||
if (b.len < n) { n = b.len; };
|
||||
// equal — true iff `a` and `b` have the same length and contents.
|
||||
// ref/hare/bytes/equal.ha:9.
|
||||
export fn equal(a: []u8, b: []u8) bool = {
|
||||
if (a.len != b.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < n) {
|
||||
if (a[i] != b[i]) { return (a[i]: i32) - (b[i]: i32); };
|
||||
i += 1;
|
||||
};
|
||||
return a.len - b.len;
|
||||
};
|
||||
|
||||
export fn hasprefix(s: str, p: str) bool = {
|
||||
if (p.len > s.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < p.len) {
|
||||
if (s[i] != p[i]) { return false; };
|
||||
for (i < a.len) {
|
||||
if (a[i] != b[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
export fn hassuffix(s: str, suf: str) bool = {
|
||||
if (suf.len > s.len) { return false; };
|
||||
let off: i32 = s.len - suf.len;
|
||||
let i: i32 = 0;
|
||||
for (i < suf.len) {
|
||||
if (s[off + i] != suf[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
// byteindex — first byte position of `needle` in `s`. Mirrors Hare's
|
||||
// strings::byteindex: a single-codepoint rune scans for the byte that
|
||||
// encodes it (ASCII only here — multi-byte UTF-8 awaits utf8 encode),
|
||||
// a str needle scans for the substring. Returns void if absent.
|
||||
export fn byteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
// index — first offset of `needle` in `s`. u8 needle scans for the
|
||||
// byte; []u8 needle scans for the substring. void if absent.
|
||||
// ref/hare/bytes/index.ha:6.
|
||||
export fn index(s: []u8, needle: (u8 | []u8)) (i32 | void) = {
|
||||
match (needle) {
|
||||
case let r: rune => {
|
||||
let c: u8 = r: u8;
|
||||
case let c: u8 => {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i] == c) { return i; };
|
||||
@@ -912,7 +894,7 @@ export fn byteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
};
|
||||
return;
|
||||
};
|
||||
case let sub: str => {
|
||||
case let sub: []u8 => {
|
||||
if (sub.len == 0) { return 0; };
|
||||
if (sub.len > s.len) { return; };
|
||||
let last: i32 = s.len - sub.len;
|
||||
@@ -933,87 +915,12 @@ export fn byteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
return;
|
||||
};
|
||||
|
||||
// contains — true iff `sub` appears in `s`. Mirrors Hare's
|
||||
// strings::contains shape (byte-wise on the str-needle case).
|
||||
export fn contains(s: str, sub: str) bool = {
|
||||
let r: (i32 | void) = byteindex(s, sub);
|
||||
match (r) {
|
||||
case let i: i32 => return true;
|
||||
case void => return false;
|
||||
};
|
||||
return false;
|
||||
};
|
||||
|
||||
// concat — joins two strings into a fresh str. Caller owns the
|
||||
// returned str's storage; release via `os.free(r.ptr, r.len)`. Mirrors
|
||||
// Hare's strings::concat shape.
|
||||
export fn concat(a: str, b: str) str = {
|
||||
let total: i32 = a.len + b.len;
|
||||
let buf: *u8 = os.alloc(total: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < a.len) { buf[i] = a[i]; i += 1; };
|
||||
let j: i32 = 0;
|
||||
for (j < b.len) { buf[a.len + j] = b[j]; j += 1; };
|
||||
let r: str;
|
||||
r.ptr = buf;
|
||||
r.len = total;
|
||||
return r;
|
||||
};
|
||||
|
||||
// dup — duplicate a string into a fresh allocation. Caller owns the
|
||||
// returned str's storage; release via `os.free(r.ptr, r.len)`. Mirrors
|
||||
// Hare's strings::dup shape — Hare returns `(str | nomem)`, ww doesn't
|
||||
// have nomem (os.alloc aborts on OOM), so we return plain `str`.
|
||||
//
|
||||
// Empty input yields a `{nil, 0}` str — Hare returns the static empty
|
||||
// string; same observable result.
|
||||
export fn dup(s: str) str = {
|
||||
let r: str;
|
||||
r.ptr = nil;
|
||||
r.len = 0;
|
||||
if (s.len == 0) { return r; };
|
||||
let buf: *u8 = os.alloc(s.len: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) { buf[i] = s[i]; i += 1; };
|
||||
r.ptr = buf;
|
||||
r.len = s.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// freeall — release every str element in `s` (those that were
|
||||
// individually allocated) plus the slice's backing storage. Mirrors
|
||||
// Hare's strings::freeall — the natural disposer for any function
|
||||
// returning a fresh `[]str` of dup'd elements (e.g. shlex.split).
|
||||
//
|
||||
// Each element is freed via os.free at its own length; the slice
|
||||
// header storage is freed at `cap * 16` bytes (one str = 16B). Empty
|
||||
// elements (`{nil, 0}` from a zero-length dup) are skipped — calling
|
||||
// os.free on a nil pointer at len 0 would tickle the rt_free guard
|
||||
// that the runtime treats as a logic bug.
|
||||
//
|
||||
// `cap == 0` means the slice was never grown (empty `[]str` with no
|
||||
// backing allocation); skip the header free in that case too.
|
||||
export fn freeall(s: []str) void = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i].len > 0) {
|
||||
os.free(s[i].ptr: *void, s[i].len: u64);
|
||||
};
|
||||
i += 1;
|
||||
};
|
||||
if (s.cap > 0) {
|
||||
os.free(s.ptr: *void, (s.cap: u64) * 16u64);
|
||||
};
|
||||
};
|
||||
|
||||
// rbyteindex — last byte position of `needle` in `s`. Mirrors Hare's
|
||||
// strings::rbyteindex. Rune needle scans for the byte that encodes it
|
||||
// (ASCII only); str needle scans for the substring. Empty str needle
|
||||
// matches at s.len.
|
||||
export fn rbyteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
// rindex — last offset of `needle` in `s`. Empty []u8 needle returns
|
||||
// s.len (ref/hare/bytes/index.ha:103 — Hare's loop yields r-0 at i=0).
|
||||
// ref/hare/bytes/index.ha:86.
|
||||
export fn rindex(s: []u8, needle: (u8 | []u8)) (i32 | void) = {
|
||||
match (needle) {
|
||||
case let r: rune => {
|
||||
let c: u8 = r: u8;
|
||||
case let c: u8 => {
|
||||
let i: i32 = s.len - 1;
|
||||
for (i >= 0) {
|
||||
if (s[i] == c) { return i; };
|
||||
@@ -1021,7 +928,7 @@ export fn rbyteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
};
|
||||
return;
|
||||
};
|
||||
case let sub: str => {
|
||||
case let sub: []u8 => {
|
||||
if (sub.len == 0) { return s.len; };
|
||||
if (sub.len > s.len) { return; };
|
||||
let i: i32 = s.len - sub.len;
|
||||
@@ -1041,9 +948,523 @@ export fn rbyteindex(s: str, needle: (str | rune)) (i32 | void) = {
|
||||
return;
|
||||
};
|
||||
|
||||
// sub — borrowed substring `s[start..end]`. Mirrors Hare's
|
||||
// strings::sub. Caller must ensure 0 <= start <= end <= s.len; out-of-
|
||||
// range indices are clamped silently here, where Hare aborts.
|
||||
// contains — true iff `needle` (byte or sub-slice) appears in `s`.
|
||||
// ref/hare/bytes/contains.ha:5 (variadic subset; see header note).
|
||||
export fn contains(s: []u8, needle: (u8 | []u8)) bool = {
|
||||
match (index(s, needle)) {
|
||||
case let i: i32 => return true;
|
||||
case void => return false;
|
||||
};
|
||||
return false;
|
||||
};
|
||||
|
||||
// hasprefix — true iff `s` starts with `pre`.
|
||||
// ref/hare/bytes/contains.ha:21.
|
||||
export fn hasprefix(s: []u8, pre: []u8) bool = {
|
||||
if (pre.len > s.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < pre.len) {
|
||||
if (s[i] != pre[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
// hassuffix — true iff `s` ends with `suf`.
|
||||
// ref/hare/bytes/contains.ha:35.
|
||||
export fn hassuffix(s: []u8, suf: []u8) bool = {
|
||||
if (suf.len > s.len) { return false; };
|
||||
let off: i32 = s.len - suf.len;
|
||||
let i: i32 = 0;
|
||||
for (i < suf.len) {
|
||||
if (s[off + i] != suf[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
// reverse — in-place reverse of `s`. ref/hare/bytes/reverse.ha:5.
|
||||
export fn reverse(s: []u8) void = {
|
||||
let i: i32 = 0;
|
||||
let j: i32 = s.len - 1;
|
||||
for (i < j) {
|
||||
let t: u8 = s[i];
|
||||
s[i] = s[j];
|
||||
s[j] = t;
|
||||
i += 1;
|
||||
j -= 1;
|
||||
};
|
||||
};
|
||||
|
||||
// zero — set every byte of `s` to 0. ref/hare/bytes/zero.ha:5.
|
||||
export fn zero(s: []u8) void = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
s[i] = 0u8;
|
||||
i += 1;
|
||||
};
|
||||
};
|
||||
|
||||
// MODULE: utf8
|
||||
// encoding/utf8 — UTF-8 encode/decode. Hare port; see
|
||||
// ref/hare/encoding/utf8/{types,rune,encode,decode,decodetable}.ha.
|
||||
//
|
||||
// The decoder is Hoehrmann's branchless DFA, originally published
|
||||
// at <https://bjoern.hoehrmann.de/utf-8/decoder/dfa/>. Hare's
|
||||
// ref/hare/encoding/utf8/decodetable.ha:4 restructures Hoehrmann's
|
||||
// flat table to 2D `[8][256]i8`; we flatten back to 1D `[2048]i8`
|
||||
// because ww cgen does not yet ship 2D arrays (task #20).
|
||||
//
|
||||
// Surface deviation from ref/hare/encoding/utf8:
|
||||
//
|
||||
// - `encoderune` takes a caller-supplied `out: []u8` and returns
|
||||
// the byte count. Hare returns a slice into a `static let buf`;
|
||||
// the caller-buffer form mirrors lib/encoding/hex.encode and
|
||||
// skips the static-buffer/slice-return pair.
|
||||
//
|
||||
// Deferred (no in-tree caller, follow-up tasks): `prev`, `slice`,
|
||||
// `position`, `remaining`, `appendrune`, `strencode`, `strdecode`.
|
||||
// Hare's string-iteration surface (`strings::iterator`/`strings::next`
|
||||
// — ref/hare/strings/iter.ha) lives under lib/strings, not here.
|
||||
|
||||
// ref/hare/encoding/utf8/types.ha:6 — incomplete trailing sequence.
|
||||
// Plain `void` (not `!void`): a truncated tail is a control-flow
|
||||
// signal, not an error caller can ignore.
|
||||
export type more = void;
|
||||
|
||||
// ref/hare/encoding/utf8/types.ha:9 — invalid UTF-8 sequence.
|
||||
export type invalid = !void;
|
||||
|
||||
// `done` is not a built-in singleton in ww (Hare ships it as part of
|
||||
// the type system). Plain `void` (not `!void`): end-of-input is a
|
||||
// continuation signal, not an error. lib/io spells its EOF the same
|
||||
// way (lib/io/io.ww:8-11).
|
||||
export type done = void;
|
||||
|
||||
// ref/hare/encoding/utf8/decodetable.ha:4 — Hoehrmann's UTF-8 DFA,
|
||||
// flat 1D `[2048]i8`. Layout: dfa[state*256 + byte] gives the next
|
||||
// state (>0), the accept transition (0 — emit rune), or invalid (-1).
|
||||
// Values match ref/hare/encoding/utf8/decodetable.ha verbatim.
|
||||
let dfa: [2048]i8 = [
|
||||
// state 0 — initial byte: ASCII accepts (0), continuation/illegal
|
||||
// byte rejects (-1), legal multibyte start emits a state.
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
3i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 4i8, 2i8, 2i8,
|
||||
5i8, 6i8, 6i8, 6i8, 7i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 1 — expecting one continuation byte (0x80..0xBF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8, 0i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 2 — expecting one continuation byte (full 0x80..0xBF range).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 3 — first byte was 0xE0; continuation byte must be 0xA0..0xBF
|
||||
// (rejects overlong 3-byte encodings).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 4 — first byte was 0xED; continuation byte must be 0x80..0x9F
|
||||
// (rejects UTF-16 surrogate codepoints U+D800..U+DFFF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8, 1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 5 — first byte was 0xF0; continuation byte must be 0x90..0xBF
|
||||
// (rejects overlong 4-byte encodings).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 6 — middle continuation byte of a 4-byte sequence (0x80..0xBF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
|
||||
// state 7 — first byte was 0xF4; continuation byte must be 0x80..0x8F
|
||||
// (rejects codepoints above U+10FFFF).
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8, 2i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
-1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8, -1i8,
|
||||
];
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:17 — payload-bit masks. Hare's
|
||||
// [2][8]u8 flattened to 1D [16]u8; row 0 (offsets 0..7) is the
|
||||
// continuation-byte mask (always 0x3F), row 1 (offsets 8..15) is the
|
||||
// initial-byte payload mask indexed by the transition class.
|
||||
let masks: [16]u8 = [
|
||||
0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8, 0x3fu8,
|
||||
0x7fu8, 0x1fu8, 0x0fu8, 0x0fu8, 0x0fu8, 0x07u8, 0x07u8, 0x07u8,
|
||||
];
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:6 — incremental decoder state.
|
||||
export type decoder = struct {
|
||||
offs: i32,
|
||||
src: []u8,
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:12.
|
||||
export fn decode(src: []u8) decoder = {
|
||||
let d: decoder;
|
||||
d.src = src;
|
||||
d.offs = 0;
|
||||
return d;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:27. Returns the next rune from a
|
||||
// decoder, `done` at end-of-input, `more` on truncated trailing
|
||||
// sequence, `invalid` on malformed input (overlong, surrogate,
|
||||
// out-of-range, bad continuation).
|
||||
//
|
||||
// Algorithm is verbatim Hoehrmann (see file header). One structural
|
||||
// rewrite: Hare encodes the "initial vs continuation byte" decision
|
||||
// as the branchless `(state - 1): uint >> 31`, which assumes a 32-bit
|
||||
// uint. ww's uint is 64-bit (cmd/wcc/type.c:58), so the shift answer
|
||||
// would be 0x1_ffff_ffff rather than 1. We spell the same predicate
|
||||
// with an explicit conditional.
|
||||
export fn next(d: *decoder) (rune | done | more | invalid) = {
|
||||
if (d.offs == d.src.len) {
|
||||
let dn: done; return dn;
|
||||
};
|
||||
let nx: i32 = 0;
|
||||
let state: i32 = 0;
|
||||
let r: u32 = 0u32;
|
||||
for (d.offs < d.src.len) {
|
||||
let b: u8 = d.src[d.offs];
|
||||
let bi: i32 = b: i32;
|
||||
let row: i32 = state * 256 + bi;
|
||||
let cell: i8 = dfa[row];
|
||||
nx = cell: i32;
|
||||
let mi: i32 = 0;
|
||||
if (state == 0) { mi = 1; };
|
||||
let m: u8 = masks[mi * 8 + (nx & 7)];
|
||||
r = (r << 6u32) | ((b & m): u32);
|
||||
if (nx <= 0) {
|
||||
d.offs += 1;
|
||||
if (nx == 0) { return r: rune; };
|
||||
let e: invalid; return e;
|
||||
};
|
||||
state = nx;
|
||||
d.offs += 1;
|
||||
};
|
||||
let mr: more; return mr;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/decode.ha:207. Strict whole-input check.
|
||||
// The hot path: tight DFA loop, no rune assembly. Bails the moment
|
||||
// the table returns -1 so malformed inputs don't pay for the rest
|
||||
// of the buffer.
|
||||
export fn validate(src: []u8) (void | invalid) = {
|
||||
let state: i32 = 0;
|
||||
let i: i32 = 0;
|
||||
for (i < src.len) {
|
||||
if (state < 0) { break; };
|
||||
let bi: i32 = src[i]: i32;
|
||||
let cell: i8 = dfa[state * 256 + bi];
|
||||
state = cell: i32;
|
||||
i += 1;
|
||||
};
|
||||
if (state == 0) { return; };
|
||||
let e: invalid; return e;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/rune.ha:5. Encoded byte length of `r` as
|
||||
// UTF-8. Callers in ww use this to size the buffer they hand to
|
||||
// [[encoderune]]; values >0x10FFFF or negative are not legal Unicode
|
||||
// codepoints and Hare aborts on them in `encoderune` itself, so we
|
||||
// keep `runesz` infallible (matches Hare).
|
||||
export fn runesz(r: rune) i32 = {
|
||||
let ch: u32 = r: u32;
|
||||
if (ch < 128u32) { return 1; };
|
||||
if (ch < 2048u32) { return 2; };
|
||||
if (ch < 65536u32) { return 3; };
|
||||
return 4;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/rune.ha:15. Expected byte length of the
|
||||
// codepoint that starts with `c`, or `invalid` if `c` cannot start
|
||||
// a legal UTF-8 sequence. Constants written in decimal because ww
|
||||
// doesn't accept Hare's `0b1000_0000` binary syntax: 0x80=128,
|
||||
// 0xC2=194, 0xE0=224, 0xF0=240, 0xF8=248.
|
||||
export fn utf8sz(c: u8) (i32 | invalid) = {
|
||||
if (c < 128u8) { return 1; };
|
||||
if (c < 194u8) { let e: invalid; return e; };
|
||||
if (c >= 248u8) { let e: invalid; return e; };
|
||||
if (c < 224u8) { return 2; };
|
||||
if (c < 240u8) { return 3; };
|
||||
return 4;
|
||||
};
|
||||
|
||||
// ref/hare/encoding/utf8/encode.ha:7. Encode `r` into `out` (caller-
|
||||
// supplied; must hold at least [[runesz]](r) bytes) and return the
|
||||
// byte count. ABORT if `r` is a UTF-16 surrogate or above U+10FFFF —
|
||||
// same precondition Hare asserts at ref/hare/encoding/utf8/encode.ha:9.
|
||||
//
|
||||
// Surface deviation: Hare returns `[]u8` (slice into a static buf).
|
||||
// ww uses the caller-buffer form (matches lib/encoding/hex.encode);
|
||||
// caller can reuse a [4]u8 stack scratch across encodes.
|
||||
export fn encoderune(out: []u8, r: rune) i32 = {
|
||||
let ch: u32 = r: u32;
|
||||
if (ch >= 0xD800u32) {
|
||||
if (ch <= 0xDFFFu32) {
|
||||
abort("utf8.encoderune: surrogate codepoint");
|
||||
};
|
||||
};
|
||||
if (ch > 0x10FFFFu32) {
|
||||
abort("utf8.encoderune: codepoint > U+10FFFF");
|
||||
};
|
||||
|
||||
let n: i32 = 0;
|
||||
let first: u8 = 0u8;
|
||||
if (ch < 0x80u32) {
|
||||
first = 0u8; n = 1;
|
||||
} else if (ch < 0x800u32) {
|
||||
first = 0xC0u8; n = 2;
|
||||
} else if (ch < 0x10000u32) {
|
||||
first = 0xE0u8; n = 3;
|
||||
} else {
|
||||
first = 0xF0u8; n = 4;
|
||||
};
|
||||
|
||||
let v: u32 = ch;
|
||||
let i: i32 = n - 1;
|
||||
for (i > 0) {
|
||||
out[i] = ((v: u8) & 0x3Fu8) | 0x80u8;
|
||||
v = v >> 6u32;
|
||||
i -= 1;
|
||||
};
|
||||
out[0] = (v: u8) | first;
|
||||
return n;
|
||||
};
|
||||
|
||||
|
||||
// MODULE: strings
|
||||
// strings — operations over str ({ptr,len}). Hare port; see
|
||||
// ref/hare/strings/.
|
||||
//
|
||||
// Documented divergences from Hare:
|
||||
//
|
||||
// - `concat(a, b)` is 2-arg. Hare ships `concat(strs: str...)`
|
||||
// (ref/hare/strings/concat.ha:5). Blocks on task #16 (cstage
|
||||
// variadic-pack drops .len of multi-field element type). Cite
|
||||
// reverts on fix.
|
||||
// - `trim` / `ltrim` / `rtrim` take a single rune. Hare's are
|
||||
// `(exclude: rune...)` (ref/hare/strings/trim.ha:54). Same
|
||||
// blocker as concat. Hare's no-rune branch (strip whitespace)
|
||||
// is also dropped — depends on a rune set.
|
||||
// - `contains` is non-variadic. Hare's is
|
||||
// `contains(haystack, needles: (str | rune)...)`
|
||||
// (ref/hare/strings/contains.ha:9). Same blocker.
|
||||
// - `byteindex` / `rbyteindex` rune arms encode via
|
||||
// `utf8.encoderune`; the legacy impls scanned for `r: u8` (an
|
||||
// undocumented ASCII-only restriction that silently dropped
|
||||
// to the wrong byte for U+80..U+7FF and higher).
|
||||
// - `dup(s: str) str` — Hare returns `(str | nomem)`. ww's
|
||||
// `os.alloc` aborts on OOM (no `nomem` type), so we return plain
|
||||
// `str`. Empty input returns `{nil, 0}`; Hare returns the static
|
||||
// empty string — same observable result.
|
||||
|
||||
use bytes;
|
||||
use utf8;
|
||||
use os;
|
||||
|
||||
// toutf8 — borrowed []u8 view of `s`. ref/hare/strings/utf8.ha:29.
|
||||
// `cap` equals `len`; the slice does not own a separate allocation.
|
||||
export fn toutf8(s: str) []u8 = {
|
||||
let r: []u8;
|
||||
r.ptr = s.ptr;
|
||||
r.len = s.len;
|
||||
r.cap = s.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// fromutf8_unsafe — borrowed str view of `in`. Does not validate.
|
||||
// ref/hare/strings/utf8.ha:10.
|
||||
export fn fromutf8_unsafe(in: []u8) str = {
|
||||
let r: str;
|
||||
r.ptr = in.ptr;
|
||||
r.len = in.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// compare — three-way bytewise codepoint-order comparison.
|
||||
// ref/hare/strings/compare.ha:12.
|
||||
export fn compare(a: str, b: str) i32 = {
|
||||
let n: i32 = a.len;
|
||||
if (b.len < n) { n = b.len; };
|
||||
let i: i32 = 0;
|
||||
for (i < n) {
|
||||
if (a[i] != b[i]) { return (a[i]: i32) - (b[i]: i32); };
|
||||
i += 1;
|
||||
};
|
||||
return a.len - b.len;
|
||||
};
|
||||
|
||||
// dup — allocate a fresh copy of `s`. Caller releases with
|
||||
// `os.free(r.ptr, r.len: u64)`. ref/hare/strings/dup.ha:7.
|
||||
export fn dup(s: str) str = {
|
||||
let r: str;
|
||||
r.ptr = nil;
|
||||
r.len = 0;
|
||||
if (s.len == 0) { return r; };
|
||||
let buf: *u8 = os.alloc(s.len: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) { buf[i] = s[i]; i += 1; };
|
||||
r.ptr = buf;
|
||||
r.len = s.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// freeall — release each element + the slice header. The natural
|
||||
// disposer for any `[]str` of dup'd elements (e.g. shlex.split).
|
||||
// ref/hare/strings/dup.ha:38.
|
||||
//
|
||||
// Empty elements (`{nil, 0}` from a zero-length dup) are skipped:
|
||||
// os.free on a nil pointer at len 0 tickles the rt_free guard. The
|
||||
// slice header itself is freed at `cap * 16` (one str = 16B); a
|
||||
// never-grown slice (cap == 0) skips the header free.
|
||||
export fn freeall(s: []str) void = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i].len > 0) {
|
||||
os.free(s[i].ptr: *void, s[i].len: u64);
|
||||
};
|
||||
i += 1;
|
||||
};
|
||||
if (s.cap > 0) {
|
||||
os.free(s.ptr: *void, (s.cap: u64) * 16u64);
|
||||
};
|
||||
};
|
||||
|
||||
// concat — fresh allocation containing `a` then `b`. Caller releases
|
||||
// with `os.free(r.ptr, r.len: u64)`. ref/hare/strings/concat.ha:5
|
||||
// (subset: Hare's `(strs: str...)` blocks on task #16).
|
||||
export fn concat(a: str, b: str) str = {
|
||||
let total: i32 = a.len + b.len;
|
||||
let buf: *u8 = os.alloc(total: u64): *u8;
|
||||
let i: i32 = 0;
|
||||
for (i < a.len) { buf[i] = a[i]; i += 1; };
|
||||
let j: i32 = 0;
|
||||
for (j < b.len) { buf[a.len + j] = b[j]; j += 1; };
|
||||
let r: str;
|
||||
r.ptr = buf;
|
||||
r.len = total;
|
||||
return r;
|
||||
};
|
||||
|
||||
// sub — borrowed `s[start..end]`. ref/hare/strings/sub.ha:30 is
|
||||
// rune-wise; this ww form is byte-wise (no rune iterator yet, planned
|
||||
// for commit 2). Clamps out-of-range silently where Hare aborts —
|
||||
// retained for the existing getopt caller; will graduate when the
|
||||
// rune-wise form lands.
|
||||
export fn sub(s: str, start: i32, end: i32) str = {
|
||||
let lo: i32 = start;
|
||||
let hi: i32 = end;
|
||||
@@ -1056,58 +1477,145 @@ export fn sub(s: str, start: i32, end: i32) str = {
|
||||
return r;
|
||||
};
|
||||
|
||||
// trimprefix — `s` with `pre` stripped from the front, or `s`
|
||||
// unchanged if it doesn't start with `pre`. Returns a borrowed view.
|
||||
// Mirrors Hare's strings::trimprefix.
|
||||
export fn trimprefix(s: str, pre: str) str = {
|
||||
if (!hasprefix(s, pre)) { return s; };
|
||||
// runebytes — encode `r` into caller's `scratch` (must hold 4 bytes)
|
||||
// and return the borrowed slice trimmed to the encoded length. Hare
|
||||
// inlines the same shape at ref/hare/strings/index.ha:132.
|
||||
fn runebytes(scratch: []u8, r: rune) []u8 = {
|
||||
let n: i32 = utf8.encoderune(scratch, r);
|
||||
let s: []u8;
|
||||
s.ptr = scratch.ptr;
|
||||
s.len = n;
|
||||
s.cap = n;
|
||||
return s;
|
||||
};
|
||||
|
||||
// hasprefix — true iff `in` begins with `prefix`.
|
||||
// ref/hare/strings/suffix.ha:8.
|
||||
export fn hasprefix(in: str, prefix: (str | rune)) bool = {
|
||||
let scratch: [4]u8;
|
||||
let p: []u8 = match (prefix) {
|
||||
case let s: str => yield toutf8(s);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.hasprefix(toutf8(in), p);
|
||||
};
|
||||
|
||||
// hassuffix — true iff `in` ends with `suff`.
|
||||
// ref/hare/strings/suffix.ha:26.
|
||||
export fn hassuffix(in: str, suff: (str | rune)) bool = {
|
||||
let scratch: [4]u8;
|
||||
let s: []u8 = match (suff) {
|
||||
case let v: str => yield toutf8(v);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.hassuffix(toutf8(in), s);
|
||||
};
|
||||
|
||||
// byteindex — byte-wise offset of `needle` in `haystack`, or void if
|
||||
// absent. ref/hare/strings/index.ha:127. Rune arm encodes via
|
||||
// utf8.encoderune (Hare passes the encoded slice straight to
|
||||
// bytes::index).
|
||||
export fn byteindex(haystack: str, needle: (str | rune)) (i32 | void) = {
|
||||
let scratch: [4]u8;
|
||||
let n: []u8 = match (needle) {
|
||||
case let s: str => yield toutf8(s);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.index(toutf8(haystack), n);
|
||||
};
|
||||
|
||||
// rbyteindex — byte-wise offset of the last `needle` in `haystack`.
|
||||
// ref/hare/strings/index.ha:138.
|
||||
export fn rbyteindex(haystack: str, needle: (str | rune)) (i32 | void) = {
|
||||
let scratch: [4]u8;
|
||||
let n: []u8 = match (needle) {
|
||||
case let s: str => yield toutf8(s);
|
||||
case let r: rune => yield runebytes(scratch[0:4], r);
|
||||
};
|
||||
return bytes.rindex(toutf8(haystack), n);
|
||||
};
|
||||
|
||||
// contains — true iff `needle` occurs in `haystack`.
|
||||
// ref/hare/strings/contains.ha:9 (subset: Hare's variadic form
|
||||
// `(needles: (str | rune)...)` blocks on task #16).
|
||||
export fn contains(haystack: str, needle: (str | rune)) bool = {
|
||||
match (byteindex(haystack, needle)) {
|
||||
case let i: i32 => return true;
|
||||
case void => return false;
|
||||
};
|
||||
return false;
|
||||
};
|
||||
|
||||
// trimprefix — `s` with `prefix` stripped from the front, or `s`
|
||||
// unchanged if it doesn't start with `prefix`. Borrowed view.
|
||||
// ref/hare/strings/trim.ha:60.
|
||||
export fn trimprefix(input: str, prefix: str) str = {
|
||||
if (!hasprefix(input, prefix)) { return input; };
|
||||
let r: str;
|
||||
r.ptr = s.ptr + (pre.len: u64);
|
||||
r.len = s.len - pre.len;
|
||||
r.ptr = input.ptr + (prefix.len: u64);
|
||||
r.len = input.len - prefix.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// trimsuffix — `s` with `suf` stripped from the end, or `s` unchanged
|
||||
// if it doesn't end with `suf`. Returns a borrowed view. Mirrors
|
||||
// Hare's strings::trimsuffix.
|
||||
export fn trimsuffix(s: str, suf: str) str = {
|
||||
if (!hassuffix(s, suf)) { return s; };
|
||||
// trimsuffix — symmetric. ref/hare/strings/trim.ha:69.
|
||||
export fn trimsuffix(input: str, suffix: str) str = {
|
||||
if (!hassuffix(input, suffix)) { return input; };
|
||||
let r: str;
|
||||
r.ptr = s.ptr;
|
||||
r.len = s.len - suf.len;
|
||||
r.ptr = input.ptr;
|
||||
r.len = input.len - suffix.len;
|
||||
return r;
|
||||
};
|
||||
|
||||
// ltrimbyte / rtrimbyte / trimbyte — strip occurrences of a single
|
||||
// byte from the left, right, or both ends. Returns a borrowed view.
|
||||
// Hare's strings::ltrim / rtrim / trim take a rune varargs set; ww's
|
||||
// subset takes a single byte (the common ASCII case).
|
||||
export fn ltrimbyte(s: str, c: u8) str = {
|
||||
// ltrim — strip occurrences of `exclude` (encoded as UTF-8) from the
|
||||
// front. Borrowed view. ref/hare/strings/trim.ha:11 (subset: single
|
||||
// rune; Hare's `(trim: rune...)` blocks on task #16). The no-rune
|
||||
// strip-whitespace branch is omitted for the same reason.
|
||||
export fn ltrim(input: str, exclude: rune) str = {
|
||||
let scratch: [4]u8;
|
||||
let pat: []u8 = runebytes(scratch[0:4], exclude);
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
if (s[i] != c) { break; };
|
||||
i += 1;
|
||||
for (i + pat.len <= input.len) {
|
||||
let j: i32 = 0;
|
||||
let ok: bool = true;
|
||||
for (j < pat.len) {
|
||||
if (input[i + j] != pat[j]) { ok = false; j = pat.len; }
|
||||
else { j += 1; };
|
||||
};
|
||||
if (!ok) { break; };
|
||||
i += pat.len;
|
||||
};
|
||||
let r: str;
|
||||
r.ptr = s.ptr + (i: u64);
|
||||
r.len = s.len - i;
|
||||
r.ptr = input.ptr + (i: u64);
|
||||
r.len = input.len - i;
|
||||
return r;
|
||||
};
|
||||
|
||||
export fn rtrimbyte(s: str, c: u8) str = {
|
||||
let n: i32 = s.len;
|
||||
for (n > 0) {
|
||||
if (s[n - 1] != c) { break; };
|
||||
n -= 1;
|
||||
// rtrim — strip occurrences of `exclude` from the end. Borrowed view.
|
||||
// ref/hare/strings/trim.ha:32 (same subset note).
|
||||
export fn rtrim(input: str, exclude: rune) str = {
|
||||
let scratch: [4]u8;
|
||||
let pat: []u8 = runebytes(scratch[0:4], exclude);
|
||||
let n: i32 = input.len;
|
||||
for (n >= pat.len) {
|
||||
let off: i32 = n - pat.len;
|
||||
let j: i32 = 0;
|
||||
let ok: bool = true;
|
||||
for (j < pat.len) {
|
||||
if (input[off + j] != pat[j]) { ok = false; j = pat.len; }
|
||||
else { j += 1; };
|
||||
};
|
||||
if (!ok) { break; };
|
||||
n -= pat.len;
|
||||
};
|
||||
let r: str;
|
||||
r.ptr = s.ptr;
|
||||
r.ptr = input.ptr;
|
||||
r.len = n;
|
||||
return r;
|
||||
};
|
||||
|
||||
export fn trimbyte(s: str, c: u8) str = {
|
||||
return rtrimbyte(ltrimbyte(s, c), c);
|
||||
// trim — strip from both ends. ref/hare/strings/trim.ha:54.
|
||||
export fn trim(input: str, exclude: rune) str = {
|
||||
return ltrim(rtrim(input, exclude), exclude);
|
||||
};
|
||||
|
||||
// MODULE: strconv
|
||||
|
||||
Reference in New Issue
Block a user