Files
ww/lib/ascii/ascii.ww
Hojun-Cho 07fed80fab lib/ascii: add strlower/strupper
Port ref/hare/ascii/string.ha strlower/strupper as the allocating entry
points: byte-wise ASCII case fold, equivalent to Hare's rune fold since
case-folding only touches bytes <0x80 and every UTF-8 multibyte byte is
>=0x80 (passes through unchanged, length-preserving). nomem arises only
from the allocation's `?`.

strlower_buf/strupper_buf are deferred: ww has no nomem-value form or
capacity-bounded static-append to express Hare's too-small-buffer path
(#230); restore the two-tier delegation when those land.

Divergence (rule 7): the empty-input fast path returns a nil/0 str
because ww's alloc([], 0) routes through nomem, whereas Hare allocs a
zero-length buffer and zero-loops; documented at the bypass site.

Test vectors mirror Hare's @test (ABC/abc/[[[/こ/empty/aB1z). Adds
lib/ascii/asciitest.ww + test/wcc/904_ascii_run.c (registered in the
Makefile TESTS list and a build rule). Regenerates the ascii-embedding
selfhost combined.ww amalgams (#110 freshness); the wwdump amalgam also
reorders the ascii block after strings to satisfy the new import edge.
2026-06-01 09:24:59 +09:00

184 lines
4.5 KiB
Plaintext

// ascii — rune-class predicates and case folding for the ASCII range.
// Matches Hare's ascii::isdigit family (rune-taking signature). Runes
// outside 0..127 always answer `false`. The lexer hot path uses these
// inline; they are expected to inline to a couple of compares.
package ascii;
import strings;
export fn isdigit(c: rune) bool = {
if (c < 48) { return false; };
if (c > 57) { return false; };
return true;
};
export fn isupper(c: rune) bool = {
if (c < 65) { return false; };
if (c > 90) { return false; };
return true;
};
export fn islower(c: rune) bool = {
if (c < 97) { return false; };
if (c > 122) { return false; };
return true;
};
export fn isalpha(c: rune) bool = {
if (isupper(c)) { return true; };
return islower(c);
};
export fn isalnum(c: rune) bool = {
if (isalpha(c)) { return true; };
return isdigit(c);
};
// isspace — the C/Hare set: space, tab, NL, VT, FF, CR.
export fn isspace(c: rune) bool = {
if (c == 32) { return true; }; // ' '
if (c == 9) { return true; }; // '\t'
if (c == 10) { return true; }; // '\n'
if (c == 11) { return true; }; // '\v'
if (c == 12) { return true; }; // '\f'
if (c == 13) { return true; }; // '\r'
return false;
};
export fn isxdigit(c: rune) bool = {
if (isdigit(c)) { return true; };
if (c >= 65) {
if (c <= 70) { return true; }; // 'A'..'F'
};
if (c >= 97) {
if (c <= 102) { return true; }; // 'a'..'f'
};
return false;
};
// valid — `c` is in the 0..127 ASCII range.
export fn valid(c: rune) bool = {
if (c < 0) { return false; };
if (c > 127) { return false; };
return true;
};
// validstr — every byte in `s` is ASCII (0..127).
export fn validstr(s: str) bool = {
let i: i32 = 0;
for (i < s.len) {
// High-bit test rather than `> 127u8`; both cgens lower
// the bitwise form identically. The `> u8` form picks
// JA vs JG depending on signed/unsigned dispatch.
if ((s[i] & 128u8) != 0u8) { return false; };
i += 1;
};
return true;
};
// iscntrl — control chars: 0..31 and 127.
export fn iscntrl(c: rune) bool = {
if (c >= 0) { if (c <= 31) { return true; }; };
if (c == 127) { return true; };
return false;
};
// isblank — space and tab.
export fn isblank(c: rune) bool = {
if (c == 32) { return true; }; // ' '
if (c == 9) { return true; }; // '\t'
return false;
};
// isprint — printable: space through '~'.
export fn isprint(c: rune) bool = {
if (c < 32) { return false; };
if (c > 126) { return false; };
return true;
};
// isgraph — printable, non-space.
export fn isgraph(c: rune) bool = {
if (c < 33) { return false; };
if (c > 126) { return false; };
return true;
};
// ispunct — printable, non-alnum, non-space.
export fn ispunct(c: rune) bool = {
if (!isgraph(c)) { return false; };
if (isalnum(c)) { return false; };
return true;
};
// tolower / toupper — fold ASCII case. Non-letters pass through.
export fn tolower(c: rune) rune = {
if (isupper(c)) { return c + 32; };
return c;
};
export fn toupper(c: rune) rune = {
if (islower(c)) { return c - 32; };
return c;
};
// strcasecmp — three-way ASCII case-insensitive compare.
export fn strcasecmp(a: str, b: str) i32 = {
let n: i32 = a.len;
if (b.len < n) { n = b.len; };
let i: i32 = 0;
for (i < n) {
let ca: rune = tolower(a[i]: rune);
let cb: rune = tolower(b[i]: rune);
if (ca != cb) { return (ca - cb): i32; };
i += 1;
};
return a.len - b.len;
};
// strlower — ASCII-lowercased copy of s, newly allocated.
// Byte-wise fold: ASCII case-fold only touches bytes <0x80; UTF-8
// multibyte bytes are >=0x80 and pass through unchanged, so byte-wise
// equals Hare's rune fold and is length-preserving.
// _buf variants deferred — ww has no nomem-value form / static-append
// builtin; restore Hare's two-tier delegation when they land (#230).
// ref/hare/ascii/string.ha:11.
export fn strlower(s: str) (str | nomem) = {
// empty bypass: ww alloc([],0) routes through nomem; Hare allocs 0
// and zero-loops (ref/hare/ascii/string.ha)
if (s.len == 0) {
let r: str;
r.ptr = nil;
r.len = 0;
return r;
};
let buf: []u8 = alloc([], s.len: u64)?;
let i: i32 = 0;
for (i < s.len) {
buf.ptr[i] = tolower(s[i]: rune): u8;
i += 1;
};
buf.len = s.len;
return strings.frombytes(buf);
};
// strupper — ASCII-uppercased copy of s, newly allocated.
// ref/hare/ascii/string.ha:33.
export fn strupper(s: str) (str | nomem) = {
if (s.len == 0) {
let r: str;
r.ptr = nil;
r.len = 0;
return r;
};
let buf: []u8 = alloc([], s.len: u64)?;
let i: i32 = 0;
for (i < s.len) {
buf.ptr[i] = toupper(s[i]: rune): u8;
i += 1;
};
buf.len = s.len;
return strings.frombytes(buf);
};