lib/regex: #123 POSIX character classes [:class:] — charclass_map + compile/exec arms
12-entry charclass_map (str, *fn(rune) bool) + bracket [:class:] recognition (compile) + charset_class_item predicate dispatch (exec), porting ref/hare/regex/regex.ha:74-87/190-204/726-733. Table-driven tests cover all 12 classes (±), negation, composition, and the "No character class after '[:'" error path. cstage-only (C-first per the speed pivot); wwstage byte-id twin owed in the batch-converge phase.
This commit is contained in:
@@ -11,23 +11,17 @@
|
|||||||
// `^`/`$`, the `\` escape, postfix `?`/`*`/`+`, alternation `|`
|
// `^`/`$`, the `\` escape, postfix `?`/`*`/`+`, alternation `|`
|
||||||
// (jump_idxs state + find_last_groupstart/shift + the insert()
|
// (jump_idxs state + find_last_groupstart/shift + the insert()
|
||||||
// builtin) and the run_thread split/jump arms; fold 4 = bracket
|
// builtin) and the run_thread split/jump arms; fold 4 = bracket
|
||||||
// expressions `[..]` (handle_bracket + the run_thread charset arm —
|
// expressions `[..]` (handle_bracket + the run_thread charset arm);
|
||||||
// the POSIX `[[:class:]]` BODY stays loud behind charclass_map);
|
|
||||||
// fold 5a = capture groups `(`/`)` (compile arms + run_thread
|
// fold 5a = capture groups `(`/`)` (compile arms + run_thread
|
||||||
// groupstart/groupend + the add_thread capture dup + the search
|
// groupstart/groupend + the add_thread capture dup + the search
|
||||||
// capture spread); fold 5b = repetition `{m,n}` (parse_repetition +
|
// capture spread); fold 5b = repetition `{m,n}` (parse_repetition +
|
||||||
// the `{` arm + the run_thread inst_repeat arm + the search
|
// the `{` arm + the run_thread inst_repeat arm + the search
|
||||||
// rep_counters prefill). Remaining: the POSIX `[:class:]` BODY (the
|
// rep_counters prefill); fold 6 = POSIX character classes
|
||||||
// last loud surface) — and replace. They land with their folds.
|
// `[[:class:]]` (charclass_map + compile/exec arms). Remaining:
|
||||||
//
|
// replace. It lands with its fold.
|
||||||
// One fold-1 construct is held back behind a filed compiler/fidelity
|
|
||||||
// gap (see the charclass_map site below):
|
|
||||||
// - charclass_map (regex.ha:74-87) — a module-level const slice of
|
|
||||||
// (str, *fn(rune) bool) tuples. Blocked on the array-literal→slice
|
|
||||||
// element-coercion checker gap (#25; type.c:402-404 #258 borrow
|
|
||||||
// uses exact type_eq, no element decay).
|
|
||||||
package regex;
|
package regex;
|
||||||
|
|
||||||
|
import ascii;
|
||||||
import bufio;
|
import bufio;
|
||||||
import io;
|
import io;
|
||||||
import memio;
|
import memio;
|
||||||
@@ -111,18 +105,23 @@ export type charset_class_item = (str, *fn(c: rune) bool);
|
|||||||
export type charset = [](charset_lit_item | charset_range_item |
|
export type charset = [](charset_lit_item | charset_range_item |
|
||||||
charset_class_item);
|
charset_class_item);
|
||||||
|
|
||||||
// ref/hare/regex/regex.ha:74-87 — charclass_map: the const
|
// ref/hare/regex/regex.ha:74-87 — POSIX class token → ascii predicate.
|
||||||
// [](str, *fn(rune) bool) table mapping POSIX class tokens to the
|
// Inline tuple type (not charset_class_item alias) matches the Hare
|
||||||
// matching ascii predicate. DEFERRED: the array-literal→slice
|
// decl form; #124 cgen closed the cross-module &fn-in-const gap.
|
||||||
// assignability check (type.c:402-404, the #258 borrow) compares
|
const charclass_map: [](str, *fn(c: rune) bool) = [
|
||||||
// element types with exact type_eq and applies NO element coercion, so
|
(":alnum:]", &ascii.isalnum),
|
||||||
// the literal `[(":alnum:]", &ascii.isalnum), ...]` (typed
|
(":alpha:]", &ascii.isalpha),
|
||||||
// `[N](untyped_str, *fn(rune) bool)`) is rejected against the declared
|
(":blank:]", &ascii.isblank),
|
||||||
// `[](str, *fn(rune) bool)`. Minimal repro: `let xs: [](size, size) =
|
(":cntrl:]", &ascii.iscntrl),
|
||||||
// [(1, 2)];`. Reshaping to a fixed `[12](...)` array would compile but
|
(":digit:]", &ascii.isdigit),
|
||||||
// is an unfaithful workaround (CLAUDE.md rule-7), so the table — and
|
(":graph:]", &ascii.isgraph),
|
||||||
// the `import ascii;` it needs — land with the consuming fold (compile)
|
(":lower:]", &ascii.islower),
|
||||||
// once the checker gap is fixed.
|
(":print:]", &ascii.isprint),
|
||||||
|
(":punct:]", &ascii.ispunct),
|
||||||
|
(":space:]", &ascii.isspace),
|
||||||
|
(":upper:]", &ascii.isupper),
|
||||||
|
(":xdigit:]", &ascii.isxdigit),
|
||||||
|
];
|
||||||
|
|
||||||
// ref/hare/regex/regex.ha:89-93.
|
// ref/hare/regex/regex.ha:89-93.
|
||||||
export type regex = struct {
|
export type regex = struct {
|
||||||
@@ -192,18 +191,11 @@ fn shift(sl: []inst) void = {
|
|||||||
// Handles a rune inside a bracket expression, mutating the in-flight
|
// Handles a rune inside a bracket expression, mutating the in-flight
|
||||||
// charset / bracket state through the pointer params.
|
// charset / bracket state through the pointer params.
|
||||||
//
|
//
|
||||||
// ref/hare/regex/regex.ha:135-225, whole except the POSIX-class arm
|
// ref/hare/regex/regex.ha:135-225. Hare's `append(charsets, [])?` /
|
||||||
// BODY: its DETECTION (`[` + `:` peek) is verbatim, but the
|
// `append(...)?` nomem propagation drops as usual (#36, ww append
|
||||||
// charclass_map scan it guards is loud-aborted — the map itself is
|
// returns void); the discarded `strings::next` advances (ha:214-215)
|
||||||
// blocked behind the array→slice element-coercion gap (#25; see the
|
// bind to throwaway lets — a bare call statement of a tagged-returning
|
||||||
// charclass_map note above) and falling through to the literal arm
|
// fn is an unprobed shape.
|
||||||
// would silently compile `[[:alpha:]]` as a 9-literal charset. The
|
|
||||||
// skip_charclass_rest block (ha:164-170) is verbatim-dead until then:
|
|
||||||
// only the loud arm sets it; it self-activates with the POSIX fold.
|
|
||||||
// Hare's `append(charsets, [])?` / `append(...)?` nomem propagation
|
|
||||||
// drops as usual (#36, ww append returns void); the discarded
|
|
||||||
// `strings::next` advances (ha:214-215) bind to throwaway lets — a
|
|
||||||
// bare call statement of a tagged-returning fn is an unprobed shape.
|
|
||||||
fn handle_bracket(
|
fn handle_bracket(
|
||||||
insts: *[]inst,
|
insts: *[]inst,
|
||||||
r: rune,
|
r: rune,
|
||||||
@@ -264,7 +256,21 @@ fn handle_bracket(
|
|||||||
*is_charset_positive = false;
|
*is_charset_positive = false;
|
||||||
} else if (r == '[' && !(peek1 is utf8.done)
|
} else if (r == '[' && !(peek1 is utf8.done)
|
||||||
&& (peek1 as rune) == ':') { // regex.ha:190-204
|
&& (peek1 as rune) == ':') { // regex.ha:190-204
|
||||||
abort("regex: POSIX character class not yet ported");
|
let rest: str = strings.iterstr(iter);
|
||||||
|
for (let cc_idx: size = 0;
|
||||||
|
cc_idx < (len(charclass_map): size);
|
||||||
|
cc_idx += 1) {
|
||||||
|
if (strings.hasprefix(rest, charclass_map[cc_idx].0)) {
|
||||||
|
let n: size = (len(*charsets): size);
|
||||||
|
append((*charsets)[n - 1],
|
||||||
|
(charclass_map[cc_idx]: charset_class_item));
|
||||||
|
*skip_charclass_rest = true;
|
||||||
|
break;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
if (!*skip_charclass_rest) {
|
||||||
|
return "No character class after '[:'": error;
|
||||||
|
};
|
||||||
} else if (is_range) { // regex.ha:205-217
|
} else if (is_range) { // regex.ha:205-217
|
||||||
let start_b: u32 = (r: u32);
|
let start_b: u32 = (r: u32);
|
||||||
let end_b: u32 = ((range_end as rune): u32);
|
let end_b: u32 = ((range_end as rune): u32);
|
||||||
@@ -1021,9 +1027,7 @@ fn run_thread(
|
|||||||
// Hare loops `for (let i = 0z; ...) match (charset[i])`
|
// Hare loops `for (let i = 0z; ...) match (charset[i])`
|
||||||
// (ha:709) — index loop with the typed-let scrutinee bind
|
// (ha:709) — index loop with the typed-let scrutinee bind
|
||||||
// (the run_thread spelling); Hare's inner `i` renames to k
|
// (the run_thread spelling); Hare's inner `i` renames to k
|
||||||
// (it shadows the thread-index param). The class arm is the
|
// (it shadows the thread-index param).
|
||||||
// loud fold boundary: compile() cannot emit a class item
|
|
||||||
// until the POSIX fold (charclass_map is #25-blocked).
|
|
||||||
for (let k: size = 0; k < (len(cset): size); k += 1) {
|
for (let k: size = 0; k < (len(cset): size); k += 1) {
|
||||||
let cur: (charset_lit_item | charset_range_item |
|
let cur: (charset_lit_item | charset_range_item |
|
||||||
charset_class_item) = cset[k];
|
charset_class_item) = cset[k];
|
||||||
@@ -1046,8 +1050,15 @@ fn run_thread(
|
|||||||
break;
|
break;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
case charset_class_item =>
|
case let class_item: charset_class_item => { // regex.ha:726-733
|
||||||
abort("regex: POSIX character class not yet ported");
|
let classfn: *fn(c: rune) bool = class_item.1;
|
||||||
|
if ((*classfn)(r)) {
|
||||||
|
// Succeeded if positive match
|
||||||
|
// Failed if negative match
|
||||||
|
matched = cs.is_positive;
|
||||||
|
break;
|
||||||
|
};
|
||||||
|
};
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
if (!matched) {
|
if (!matched) {
|
||||||
|
|||||||
@@ -14,10 +14,9 @@
|
|||||||
// Fold 2a ports compile()'s lit/any/match arms only; exec lives in
|
// Fold 2a ports compile()'s lit/any/match arms only; exec lives in
|
||||||
// later folds, so the compile_* cases pin the emitted inst PROGRAM
|
// later folds, so the compile_* cases pin the emitted inst PROGRAM
|
||||||
// (shape + payloads via indexed match-extraction), not matching.
|
// (shape + payloads via indexed match-extraction), not matching.
|
||||||
// charclass_map's fn-ptr table is deferred behind the array→slice
|
// fold-6 tests exercise POSIX character classes (charclass_map +
|
||||||
// element-coercion checker gap (see regex.ww), so this test does not
|
// compile/exec arms). Earlier folds pin variant discrimination
|
||||||
// exercise the POSIX-class predicate dispatch yet — it pins variant
|
// (including the nominally-distinct same-underlying
|
||||||
// discrimination (including the nominally-distinct same-underlying
|
|
||||||
// inst_split/inst_jump/inst_groupstart `size` aliases and the
|
// inst_split/inst_jump/inst_groupstart `size` aliases and the
|
||||||
// inst_any/inst_skip/inst_groupend `void` aliases), payload extraction,
|
// inst_any/inst_skip/inst_groupend `void` aliases), payload extraction,
|
||||||
// the regex/capture struct shapes, and finish(). Same
|
// the regex/capture struct shapes, and finish(). Same
|
||||||
@@ -1436,7 +1435,7 @@ type cerow = struct {
|
|||||||
|
|
||||||
// cssig — flatten a charset element: lits as 1_000_000 + codepoint,
|
// cssig — flatten a charset element: lits as 1_000_000 + codepoint,
|
||||||
// ranges as start*10000 + end (codepoints stay < 10000 in these rows),
|
// ranges as start*10000 + end (codepoints stay < 10000 in these rows),
|
||||||
// class items can't exist (compile()'s POSIX arm is loud).
|
// class items as -2 (presence sentinel — fn-ptr address not stable).
|
||||||
// The element binds structurally, not via the `charset` alias — an
|
// The element binds structurally, not via the `charset` alias — an
|
||||||
// alias-typed slice local's index read mis-scales in wwstage (#68).
|
// alias-typed slice local's index read mis-scales in wwstage (#68).
|
||||||
fn cssig(cs: [](charset_lit_item | charset_range_item |
|
fn cssig(cs: [](charset_lit_item | charset_range_item |
|
||||||
@@ -1448,6 +1447,7 @@ fn cssig(cs: [](charset_lit_item | charset_range_item |
|
|||||||
return 1000000 + ((l: rune): i64);
|
return 1000000 + ((l: rune): i64);
|
||||||
case let range: charset_range_item =>
|
case let range: charset_range_item =>
|
||||||
return (range.0: i64) * 10000 + (range.1: i64);
|
return (range.0: i64) * 10000 + (range.1: i64);
|
||||||
|
case charset_class_item => return -2; // fn-ptr address not stable
|
||||||
case => return -1;
|
case => return -1;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
@@ -2427,6 +2427,150 @@ type prrow = struct {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// ---- fold 6: POSIX character classes ---------------------------------
|
||||||
|
|
||||||
|
// compile error: `[[:` with no valid class name
|
||||||
|
// ref/hare/regex/+test.ha — no direct cite; error string from
|
||||||
|
// regex.ha:203 "No character class after '[:'".
|
||||||
|
@test fn fold6_compile_errors() void = {
|
||||||
|
let rows: [1]cerow = [
|
||||||
|
cerow { pat = "[[:xyz", want = "No character class after '[:'" },
|
||||||
|
];
|
||||||
|
let i: i32 = 0;
|
||||||
|
for (i < len(rows)) {
|
||||||
|
let p: str = rows[i].pat;
|
||||||
|
match (regex.compile(p)) {
|
||||||
|
case let e: regex.error => {
|
||||||
|
let w: str = rows[i].want;
|
||||||
|
if (strings.compare((e: str), w) != 0) { fail(); };
|
||||||
|
};
|
||||||
|
case => fail();
|
||||||
|
};
|
||||||
|
i += 1;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// charsets-table content pin: a POSIX class bracket emits ONE
|
||||||
|
// charset_class_item element (cssig → -2), not a literal expansion.
|
||||||
|
// ref/hare/regex/+test.ha:305-308 POSIX class rows.
|
||||||
|
@test fn fold6_charsets() void = {
|
||||||
|
// "[[:digit:]]" → 1 charset, 1 class elem
|
||||||
|
let c: (regex.regex | regex.error | nomem) = regex.compile("[[:digit:]]");
|
||||||
|
match (c) {
|
||||||
|
case let re: regex.regex => {
|
||||||
|
if (re.charsets.len != 1) { fail(); };
|
||||||
|
let cs0: [](charset_lit_item | charset_range_item |
|
||||||
|
charset_class_item) = re.charsets[0];
|
||||||
|
if ((len(cs0): i32) != 1) { fail(); };
|
||||||
|
if (cssig(cs0, 0) != -2) { fail(); };
|
||||||
|
regex.finish(&re);
|
||||||
|
};
|
||||||
|
case => fail();
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// end-to-end find/test rows for all 12 POSIX classes + negation.
|
||||||
|
// ref/hare/regex/+test.ha:278-345 POSIX subset.
|
||||||
|
@test fn fold6_find_cases() void = {
|
||||||
|
let rows: [28]fcase = [
|
||||||
|
// digit
|
||||||
|
fcase { expr = "[[:digit:]]", input = "5", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = "5" },
|
||||||
|
fcase { expr = "[[:digit:]]", input = "a", matches = false, ... },
|
||||||
|
// alpha
|
||||||
|
fcase { expr = "^[[:alpha:]]+$", input = "abc", matches = true,
|
||||||
|
start = 0, sb = 0, end = 3, eb = 3, content = "abc" },
|
||||||
|
fcase { expr = "^[[:alpha:]]+$", input = "abc1", matches = false, ... },
|
||||||
|
// alnum
|
||||||
|
fcase { expr = "^[[:alnum:]]+$", input = "abc9", matches = true,
|
||||||
|
start = 0, sb = 0, end = 4, eb = 4, content = "abc9" },
|
||||||
|
fcase { expr = "^[[:alnum:]]+$", input = "abc!", matches = false, ... },
|
||||||
|
// space
|
||||||
|
fcase { expr = "[[:space:]]", input = " ", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = " " },
|
||||||
|
fcase { expr = "[[:space:]]", input = "a", matches = false, ... },
|
||||||
|
// upper
|
||||||
|
fcase { expr = "^[[:upper:]]+$", input = "ABC", matches = true,
|
||||||
|
start = 0, sb = 0, end = 3, eb = 3, content = "ABC" },
|
||||||
|
fcase { expr = "^[[:upper:]]+$", input = "ABc", matches = false, ... },
|
||||||
|
// lower
|
||||||
|
fcase { expr = "^[[:lower:]]+$", input = "abc", matches = true,
|
||||||
|
start = 0, sb = 0, end = 3, eb = 3, content = "abc" },
|
||||||
|
fcase { expr = "^[[:lower:]]+$", input = "abC", matches = false, ... },
|
||||||
|
// xdigit
|
||||||
|
fcase { expr = "^[[:xdigit:]]+$", input = "0aF", matches = true,
|
||||||
|
start = 0, sb = 0, end = 3, eb = 3, content = "0aF" },
|
||||||
|
fcase { expr = "^[[:xdigit:]]+$", input = "0g", matches = false, ... },
|
||||||
|
// blank (space or tab)
|
||||||
|
fcase { expr = "[[:blank:]]", input = "\t", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = "\t" },
|
||||||
|
fcase { expr = "[[:blank:]]", input = "a", matches = false, ... },
|
||||||
|
// punct
|
||||||
|
fcase { expr = "[[:punct:]]", input = ".", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = "." },
|
||||||
|
fcase { expr = "[[:punct:]]", input = "a", matches = false, ... },
|
||||||
|
// graph
|
||||||
|
fcase { expr = "[[:graph:]]", input = "!", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = "!" },
|
||||||
|
fcase { expr = "[[:graph:]]", input = " ", matches = false, ... },
|
||||||
|
// print
|
||||||
|
fcase { expr = "[[:print:]]", input = " ", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = " " },
|
||||||
|
fcase { expr = "[[:print:]]", input = "\x01", matches = false, ... },
|
||||||
|
// cntrl
|
||||||
|
fcase { expr = "[[:cntrl:]]", input = "\x01", matches = true,
|
||||||
|
start = 0, sb = 0, end = 1, eb = 1, content = "\x01" },
|
||||||
|
fcase { expr = "[[:cntrl:]]", input = "a", matches = false, ... },
|
||||||
|
// negated class
|
||||||
|
fcase { expr = "^[^[:digit:]]+$", input = "abc", matches = true,
|
||||||
|
start = 0, sb = 0, end = 3, eb = 3, content = "abc" },
|
||||||
|
fcase { expr = "^[^[:digit:]]+$", input = "ab5", matches = false, ... },
|
||||||
|
// composition: digit+ in surrounding text
|
||||||
|
fcase { expr = "[[:digit:]]+", input = "abc123def",
|
||||||
|
matches = true, start = 3, sb = 3, end = 6, eb = 6,
|
||||||
|
content = "123" },
|
||||||
|
fcase { expr = "[[:digit:]]+", input = "nodigits",
|
||||||
|
matches = false, ... },
|
||||||
|
];
|
||||||
|
let i: i32 = 0;
|
||||||
|
for (i < len(rows)) {
|
||||||
|
let ex: str = rows[i].expr;
|
||||||
|
let inp: str = rows[i].input;
|
||||||
|
let c: (regex.regex | regex.error | nomem) = regex.compile(ex);
|
||||||
|
match (c) {
|
||||||
|
case let re: regex.regex => {
|
||||||
|
let fr: (regex.result | nomem) = regex.find(&re, inp);
|
||||||
|
if (!(fr is regex.result)) { fail(); };
|
||||||
|
let res: regex.result = fr as regex.result;
|
||||||
|
if (rows[i].matches) {
|
||||||
|
if (len(res) != 1) { fail(); };
|
||||||
|
if (res[0].start != rows[i].start) { fail(); };
|
||||||
|
if (res[0].start_bytesize != rows[i].sb) {
|
||||||
|
fail();
|
||||||
|
};
|
||||||
|
if (res[0].end != rows[i].end) { fail(); };
|
||||||
|
if (res[0].end_bytesize != rows[i].eb) {
|
||||||
|
fail();
|
||||||
|
};
|
||||||
|
let wc: str = rows[i].content;
|
||||||
|
if (strings.compare(res[0].content, wc) != 0) {
|
||||||
|
fail();
|
||||||
|
};
|
||||||
|
} else {
|
||||||
|
if (len(res) != 0) { fail(); };
|
||||||
|
};
|
||||||
|
let tr: (bool | nomem) = regex.test(&re, inp);
|
||||||
|
if (!(tr is bool)) { fail(); };
|
||||||
|
if ((tr as bool) != (len(res) != 0)) { fail(); };
|
||||||
|
regex.result_free(res);
|
||||||
|
regex.finish(&re);
|
||||||
|
};
|
||||||
|
case => fail();
|
||||||
|
};
|
||||||
|
i += 1;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
export fn main() i32 = {
|
export fn main() i32 = {
|
||||||
signalled = 1; lit_and_match();
|
signalled = 1; lit_and_match();
|
||||||
signalled = 2; size_aliases_distinct();
|
signalled = 2; size_aliases_distinct();
|
||||||
@@ -2473,5 +2617,8 @@ export fn main() i32 = {
|
|||||||
signalled = 44; fold5b_compile_errors();
|
signalled = 44; fold5b_compile_errors();
|
||||||
signalled = 45; fold5b_find_cases();
|
signalled = 45; fold5b_find_cases();
|
||||||
signalled = 46; fold5b_findall();
|
signalled = 46; fold5b_findall();
|
||||||
|
signalled = 47; fold6_compile_errors();
|
||||||
|
signalled = 48; fold6_charsets();
|
||||||
|
signalled = 49; fold6_find_cases();
|
||||||
return 0;
|
return 0;
|
||||||
};
|
};
|
||||||
|
|||||||
Reference in New Issue
Block a user