Files
ww/selfhost/cmd/w6a/lex.ww
Hojun-Cho 559b77db40 selfhost/cmd/w6a: align parsenum to strtoll(base 0) semantics (#62)
w6a's parsenum diverged from the C twin's strtoll(s,end,0)
(cmd/w6a/lex.c:30) on three hand-written-asm edge shapes (all
gate-blind — w6c emits the canonical $5/$8/-8(BP), never these):
  (a) `$ 5`  — leading whitespace: strtoll skips it (->5); ww had no
              skip and silently encoded imm 0.
  (b) `$08`  — strtoll base-0 reads a leading 0 as octal, stops at '8'
              (->0); ww parsed it as decimal 8.
  (c) `-(BP)` — strtoll/cstage require a digit after the sign, so a bare
              `-(` is unrecognised operand; ww silently took it as 0(BP).
Add the whitespace skip + octal base-0 detection to parsenum, and the
digit-after-sign guard to the operand scanner — both assemblers now
agree byte-for-byte (a/b) and both reject (c).

Not a Hare item (w6a is ww's plan9-lineage assembler); reference is the
C strtoll twin. w6a embeds into its own combined.ww snapshot; regen'd.
530_w6a_parsenum pins the byte-identity + both-reject matrix.
2026-06-13 11:03:42 +09:00

78 lines
2.4 KiB
Plaintext

// selfhost/cmd/w6a/lex.ww — port of cmd/w6a/lex.c.
//
// Character-level helpers for w6a's line-oriented parser. The parser
// itself is in parse.ww; here we keep tokenisers for identifiers and
// numbers so parse.ww stays focused on syntax.
package w6a;
export fn isidstart(c: i32) bool = {
if (c == 95) { return true; };
if (c >= 65) { if (c <= 90) { return true; }; }; // A-Z
if (c >= 97) { if (c <= 122) { return true; }; }; // a-z
return false;
};
export fn isidcont(c: i32) bool = {
if (isidstart(c)) { return true; };
if (c >= 48) { if (c <= 57) { return true; }; }; // 0-9
if (c == 46) { return true; }; // .
return false;
};
// parsenum — read a leading [+-]?[0x|0X|0]?digits from p[0..n-1].
// Returns (value, consumed). Stops at first non-digit.
// Plain Plan 9-style: $123 / $0x1f / $-7. Decimal default; 0x prefix
// for hex; 0 prefix for octal when followed by a digit (else just 0).
export fn parsenum(p: *u8, n: u64) (i64, u64) = {
// strtoll(s, end, 0) semantics, matching the C twin cmd/w6a/lex.c:30:
// skip leading whitespace, optional sign, base-0 prefix detection
// (0x -> hex, leading 0 -> octal, else decimal). w6c never emits the
// `$ 5` / `$08` edge shapes; this aligns the hand-written-asm path
// with cstage so the two assemblers agree byte-for-byte (#62).
let i: u64 = 0u64;
for (i < n) {
if (p[i] != 32u8) { if (p[i] != 9u8) { break; }; };
i += 1u64;
};
let neg: bool = false;
if (i < n) {
if (p[i] == 45u8) { neg = true; i += 1u64; }
else { if (p[i] == 43u8) { i += 1u64; }; };
};
let base: i64 = 10i64;
if (i < n) {
if (p[i] == 48u8) { // leading '0' -> octal, unless '0x'/'0X'
if (i + 1u64 < n) {
if (p[i + 1u64] == 120u8) { base = 16i64; i += 2u64; }
else { if (p[i + 1u64] == 88u8) { base = 16i64; i += 2u64; }
else { base = 8i64; i += 1u64; }; };
} else { base = 8i64; i += 1u64; };
};
};
let v: i64 = 0i64;
let scan: bool = true;
for (scan) {
if (i >= n) { scan = false; }
else {
let c: u8 = p[i];
let d: i64 = -1i64;
if (c >= 48u8) { if (c <= 57u8) { d = (c - 48u8): i64; }; };
if (d < 0i64) {
if (base == 16i64) {
if (c >= 97u8) { if (c <= 102u8) { d = (c - 97u8): i64 + 10i64; }; };
if (c >= 65u8) { if (c <= 70u8) { d = (c - 65u8): i64 + 10i64; }; };
};
};
if (d < 0i64) { scan = false; }
else { if (d >= base) { scan = false; }
else {
v = v * base + d;
i += 1u64;
}; };
};
};
if (neg) { v = -v; };
return v, i;
};