Files
ww/cmd/wcc/lex.c
Hojun-Cho 0197dfb9e6 lex,wwi: \u/\U unicode escapes + wide-rune .wwi round-trip, both stages (#50)
Hare-faithful \u (4 hex) / \U (8 hex) escapes; \x/\u/\U share one codepoint
path (ref/hare/hare/lex/lex.ha lex_unicode); string literals UTF-8-encode
multi-byte codepoints (cstage inline utf8enc, wwstage utf8.encoderune). The
.wwi producer rune serializer now emits \u/\U so exported wide-rune defs
round-trip (was a fatal >0xFF). Reject >0x10FFFF and surrogates with Hare-
verbatim error strings. Closes the int-cast spelling divergence (#48 RUNE_MAX).
Both stages byte-identical; 446 tests pass.
2026-06-18 22:47:53 +09:00

631 lines
15 KiB
C

/*
* lex.c — hand-rolled DFA. UTF-8 source, ASCII operators.
*
* Comments: //... and (slash-star ... star-slash). Both stripped.
* Whitespace: space, tab, CR, NL.
* Identifiers: [A-Za-z_][A-Za-z0-9_]* — also matches keywords; we
* look up the kw table after lexing the run.
* Integer: 0x[0-9a-fA-F_]+, 0o[0-7_]+, 0b[01_]+, [0-9][0-9_]*
* Float: [0-9]+'.'[0-9]+([eE][+-]?[0-9]+)?
* Rune: 'x' with C-like escapes
* String: "..." with C-like escapes
* Operators: longest match.
*
* No automatic semicolon insertion (Hare rule). The lexer only emits
* what is in the source; the parser is responsible for non-empty rules.
*/
#include "ww.h"
#include <stdlib.h>
#include <string.h>
#include <errno.h>
void
lexinit(Lex *l, Arena *a, const char *file, const char *src, u64 len)
{
memset(l, 0, sizeof *l);
l->file = file;
l->src = src;
l->srclen = len;
l->line = 1;
l->col = 1;
l->a = a;
}
static int
lpeek(Lex *l, u64 ahead)
{
u64 p = l->pos + ahead;
if (p >= l->srclen)
return -1;
return (unsigned char)l->src[p];
}
static int
lget(Lex *l)
{
if (l->pos >= l->srclen)
return -1;
int c = (unsigned char)l->src[l->pos++];
if (c == '\n') {
l->line++;
l->col = 1;
} else {
l->col++;
}
return c;
}
static Pos
lpos(Lex *l)
{
Pos p = { l->file, l->line, l->col };
return p;
}
static int
isidstart(int c)
{
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || c == '_';
}
static int
isidcont(int c)
{
return isidstart(c) || (c >= '0' && c <= '9');
}
static int
ishex(int c)
{
return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') ||
(c >= 'A' && c <= 'F');
}
/* skip whitespace and comments. returns 0 on EOF, else 1. */
static int
skipws(Lex *l)
{
for (;;) {
int c = lpeek(l, 0);
if (c < 0)
return 0;
if (c == ' ' || c == '\t' || c == '\r' || c == '\n') {
lget(l);
continue;
}
if (c == '/' && lpeek(l, 1) == '/') {
lget(l); lget(l); /* consume '//' */
/* #16 option-B: the driver emits `//ww:module-reset`
* before a package-less file's bytes; recognize the
* whole-line directive (without consuming differently)
* and flag it so lexnext emits TK_MODRESET. The body is
* then skipped like any comment. Mirrors the removed
* `// MODULE:` lexer directive. */
{
static const char pre[] = "ww:module";
size_t i = 0;
while (pre[i] && lpeek(l, i) == pre[i])
i++;
if (pre[i] == '\0') {
int nx = lpeek(l, i);
if (nx == '-') {
static const char rest[] = "-reset";
size_t j = 0;
while (rest[j] && lpeek(l, i + j) == rest[j])
j++;
if (rest[j] == '\0') {
int af = lpeek(l, i + j);
if (af == '\n' || af < 0) {
l->modreset = 1;
} else if (af == ' '
|| af == '\t') {
/* `//ww:module-reset <path>`
* — sep primary body tagged by
* its full dotted import path so
* definer == importer (#57). */
size_t k = i + j;
while (lpeek(l, k) == ' '
|| lpeek(l, k) == '\t')
k++;
size_t s = k;
int dch;
while ((dch = lpeek(l, k)) >= 0
&& dch != '\n'
&& dch != '\r'
&& dch != ' '
&& dch != '\t')
k++;
l->modreset = 1;
if (k > s)
l->modresetpath =
astrndup(l->a,
l->src + l->pos + s,
k - s);
}
}
} else if (nx == ' ' || nx == '\t') {
/* `//ww:module <path>` — M1 import boundary. */
size_t k = i;
while (lpeek(l, k) == ' '
|| lpeek(l, k) == '\t')
k++;
size_t s = k;
int ch;
while ((ch = lpeek(l, k)) >= 0
&& ch != '\n' && ch != '\r'
&& ch != ' ' && ch != '\t')
k++;
if (k > s)
l->modpath = astrndup(l->a,
l->src + l->pos + s, k - s);
}
}
}
while ((c = lpeek(l, 0)) >= 0 && c != '\n')
lget(l);
continue;
}
if (c == '/' && lpeek(l, 1) == '*') {
lget(l); lget(l);
int prev = -1;
for (;;) {
int x = lget(l);
if (x < 0) {
Pos p = lpos(l);
errorf(p, "unterminated /* comment");
l->errs++;
return 0;
}
if (prev == '*' && x == '/')
break;
prev = x;
}
continue;
}
return 1;
}
}
static u64
parseint(const char *s, u64 n, int base, int *ok)
{
u64 v = 0;
int got = 0;
for (u64 i = 0; i < n; i++) {
int c = (unsigned char)s[i];
if (c == '_')
continue;
int d;
if (c >= '0' && c <= '9') d = c - '0';
else if (c >= 'a' && c <= 'f') d = c - 'a' + 10;
else if (c >= 'A' && c <= 'F') d = c - 'A' + 10;
else { *ok = 0; return 0; }
if (d >= base) { *ok = 0; return 0; }
/* overflow? cheap check */
if (v > (u64)~0ULL / (u64)base) { *ok = 0; return 0; }
v = v * (u64)base + (u64)d;
got = 1;
}
*ok = got;
return v;
}
/* shared escape decoder for \xHH (n=2), \uHHHH (n=4), \UHHHHHHHH (n=8).
* All three yield a codepoint, not a raw byte — mirrors
* ref/hare/hare/lex/lex.ha:347 fn lex_unicode. Error strings copied
* verbatim from that reference for diagnostic fidelity (#50). */
static int
lexunicode(Lex *l, int n, int *out)
{
u32 u = 0;
for (int i = 0; i < n; i++) {
int c = lget(l);
if (c < 0) {
Pos p = lpos(l);
errorf(p, "unexpected EOF scanning for escape");
l->errs++;
return -1;
}
if (!ishex(c)) {
Pos p = lpos(l);
errorf(p, "unexpected rune scanning for escape");
l->errs++;
return -1;
}
int d = (c <= '9' ? c - '0' : (c | 0x20) - 'a' + 10);
u = (u << 4) | (u32)d;
}
if (u > 0x10FFFF || (u >= 0xD800 && u < 0xE000)) {
Pos p = lpos(l);
errorf(p, "invalid unicode codepoint in escape");
l->errs++;
return -1;
}
*out = (int)u;
return 0;
}
/* utf8enc — encode codepoint cp (already validated <= 0x10FFFF and
* non-surrogate by lexunicode) into out (>= 4 bytes), return the byte
* count. The C bootstrap has no stdlib; this mirrors lib/encoding/utf8
* encoderune (the ww side calls that directly). */
static int
utf8enc(u32 cp, char *out)
{
if (cp < 0x80) {
out[0] = (char)cp;
return 1;
} else if (cp < 0x800) {
out[0] = (char)(0xC0 | (cp >> 6));
out[1] = (char)(0x80 | (cp & 0x3F));
return 2;
} else if (cp < 0x10000) {
out[0] = (char)(0xE0 | (cp >> 12));
out[1] = (char)(0x80 | ((cp >> 6) & 0x3F));
out[2] = (char)(0x80 | (cp & 0x3F));
return 3;
}
out[0] = (char)(0xF0 | (cp >> 18));
out[1] = (char)(0x80 | ((cp >> 12) & 0x3F));
out[2] = (char)(0x80 | ((cp >> 6) & 0x3F));
out[3] = (char)(0x80 | (cp & 0x3F));
return 4;
}
static int
escape(Lex *l, int *out)
{
int c = lget(l);
if (c < 0) return -1;
switch (c) {
case 'n': *out = '\n'; return 0;
case 't': *out = '\t'; return 0;
case 'r': *out = '\r'; return 0;
case '\\': *out = '\\'; return 0;
case '\'': *out = '\''; return 0;
case '"': *out = '"'; return 0;
case '0': *out = '\0'; return 0;
case 'a': *out = '\a'; return 0;
case 'b': *out = '\b'; return 0;
case 'f': *out = '\f'; return 0;
case 'v': *out = '\v'; return 0;
case 'x': return lexunicode(l, 2, out);
case 'u': return lexunicode(l, 4, out);
case 'U': return lexunicode(l, 8, out);
}
{ Pos p = lpos(l); errorf(p, "bad escape \\%c", c); l->errs++; }
return -1;
}
static Tok
lexnum(Lex *l, Pos start)
{
Tok t = (Tok){ TK_INT, start, NULL, 0, {0}, TK_NONE };
u64 begin = l->pos;
int base = 10;
int isfloat = 0;
int c = lpeek(l, 0);
if (c == '0' && (lpeek(l, 1) == 'x' || lpeek(l, 1) == 'X')) {
lget(l); lget(l);
base = 16;
while ((c = lpeek(l, 0)) >= 0 && (ishex(c) || c == '_'))
lget(l);
} else if (c == '0' && (lpeek(l, 1) == 'b' || lpeek(l, 1) == 'B')) {
lget(l); lget(l);
base = 2;
while ((c = lpeek(l, 0)) >= 0 && (c == '0' || c == '1' || c == '_'))
lget(l);
} else if (c == '0' && (lpeek(l, 1) == 'o' || lpeek(l, 1) == 'O')) {
lget(l); lget(l);
base = 8;
while ((c = lpeek(l, 0)) >= 0 && ((c >= '0' && c <= '7') || c == '_'))
lget(l);
} else {
while ((c = lpeek(l, 0)) >= 0 && ((c >= '0' && c <= '9') || c == '_'))
lget(l);
if (lpeek(l, 0) == '.' && lpeek(l, 1) >= '0' && lpeek(l, 1) <= '9') {
isfloat = 1;
lget(l);
while ((c = lpeek(l, 0)) >= 0 && ((c >= '0' && c <= '9') || c == '_'))
lget(l);
c = lpeek(l, 0);
if (c == 'e' || c == 'E') {
lget(l);
if (lpeek(l, 0) == '+' || lpeek(l, 0) == '-')
lget(l);
while ((c = lpeek(l, 0)) >= 0 && c >= '0' && c <= '9')
lget(l);
}
}
}
u64 n = l->pos - begin;
t.text = astrndup(l->a, l->src + begin, n);
t.tlen = n;
if (isfloat) {
t.kind = TK_FLOAT;
/* strdup with underscores stripped before strtod */
char *clean = amalloc(l->a, n + 1);
u64 j = 0;
for (u64 i = 0; i < n; i++)
if (l->src[begin + i] != '_')
clean[j++] = l->src[begin + i];
clean[j] = '\0';
errno = 0;
t.v.fval = strtod(clean, NULL);
if (errno) {
errorf(start, "bad float literal '%s'", t.text);
l->errs++;
}
} else {
const char *digs = l->src + begin;
u64 dn = n;
if (base != 10) {
digs += 2;
dn -= 2;
}
int ok = 0;
t.v.uval = parseint(digs, dn, base, &ok);
if (!ok) {
errorf(start, "bad integer literal '%s'", t.text);
l->errs++;
t.kind = TK_ERR;
}
}
/* Typed suffix: i8/i16/i32/i64, u8/u16/u32/u64, f32/f64.
* Must be glued (no whitespace) to the digits. We grab the
* adjacent identifier-like run and accept it only if it's one
* of the recognised type names. */
if (isidstart(lpeek(l, 0))) {
u64 sb = l->pos;
while (isidcont(lpeek(l, 0))) lget(l);
u64 sl = l->pos - sb;
const char *names[] = {
"i8", "i16", "i32", "i64",
"u8", "u16", "u32", "u64",
"f32", "f64", NULL
};
const char *match = NULL;
for (int i = 0; names[i]; i++) {
u64 nl = strlen(names[i]);
if (nl == sl && memcmp(names[i], l->src + sb, nl) == 0) {
match = names[i];
break;
}
}
if (match) {
t.tsuffix = astrndup(l->a, l->src + sb, sl);
} else {
/* not a known suffix — rewind so the run becomes a
* separate token. */
l->pos = sb;
}
}
return t;
}
static Tok
lexident(Lex *l, Pos start)
{
u64 begin = l->pos;
while (isidcont(lpeek(l, 0)))
lget(l);
u64 n = l->pos - begin;
const char *p = l->src + begin;
/* bare '_' is the discard marker. `_x`, `_1` are normal idents. */
if (n == 1 && p[0] == '_') {
Tok t = (Tok){ TK_UNDER, start, astrndup(l->a, p, n), n, {0}, TK_NONE };
return t;
}
Tkind k = kwlookup(p, n);
Tok t = (Tok){ k != TK_NONE ? k : TK_IDENT, start,
astrndup(l->a, p, n), n, {0}, TK_NONE };
return t;
}
static Tok
lexstr(Lex *l, Pos start)
{
/* opening quote already consumed by caller */
u64 cap = 32, n = 0;
char *buf = amalloc(l->a, cap);
for (;;) {
int c = lpeek(l, 0);
if (c < 0) {
errorf(start, "unterminated string");
l->errs++;
Tok t = (Tok){ TK_ERR, start, astrndup(l->a, "", 0), 0, {0}, TK_NONE };
return t;
}
if (c == '"') { lget(l); break; }
/* Escape-decoded values are codepoints and UTF-8-encode into
* 1-4 bytes (mirrors Hare's memio::appendrune in lex_string,
* ref/hare/hare/lex/lex.ha:431). Raw source bytes are already
* UTF-8 and pass through unchanged — re-encoding them would
* double-encode the >0x7F continuation bytes. */
char enc[4];
int el;
if (c == '\\') {
int ch;
lget(l);
if (escape(l, &ch) < 0)
ch = 0;
el = utf8enc((u32)ch, enc);
} else {
enc[0] = (char)lget(l);
el = 1;
}
if (n + el >= cap) {
u64 ncap = cap * 2;
while (n + el >= ncap)
ncap *= 2;
char *nb = amalloc(l->a, ncap);
memcpy(nb, buf, n);
buf = nb;
cap = ncap;
}
for (int i = 0; i < el; i++)
buf[n++] = enc[i];
}
buf[n] = '\0';
Tok t = (Tok){ TK_STR, start, buf, n, {0}, TK_NONE };
return t;
}
static Tok
lexrune(Lex *l, Pos start)
{
int ch;
int c = lpeek(l, 0);
if (c < 0) {
errorf(start, "unterminated rune");
l->errs++;
return (Tok){ TK_ERR, start, "", 0, {0}, TK_NONE };
}
if (c == '\\') {
lget(l);
if (escape(l, &ch) < 0)
ch = 0;
} else {
ch = lget(l);
}
if (lpeek(l, 0) != '\'') {
errorf(start, "rune literal missing closing '");
l->errs++;
return (Tok){ TK_ERR, start, "", 0, {0}, TK_NONE };
}
lget(l);
Tok t = (Tok){ TK_RUNE, start, NULL, 0, {0}, TK_NONE };
t.v.uval = (u64)(u32)ch;
t.text = aprintf(l->a, "%d", ch);
t.tlen = strlen(t.text);
return t;
}
#define EMIT(K) do { Tok _t = (Tok){ (K), start, NULL, 0, {0}, TK_NONE }; \
_t.text = tokname(K); _t.tlen = strlen(_t.text); return _t; } while (0)
Tok
lexnext(Lex *l)
{
int more = skipws(l);
Pos start = lpos(l);
/* A `//ww:module-reset` seen in the skipped run surfaces as its own
* token before the next real one (#16 option-B boundary reset). */
if (l->modreset) {
l->modreset = 0;
const char *rp = l->modresetpath;
l->modresetpath = NULL;
/* path-carrying reset → text=path (#57); bare reset → text=NULL */
Tok _t = (Tok){ TK_MODRESET, start, rp, rp ? strlen(rp) : 0,
{0}, TK_NONE };
return _t;
}
if (l->modpath) {
const char *mp = l->modpath;
l->modpath = NULL;
Tok _t = (Tok){ TK_MODPATH, start, NULL, 0, {0}, TK_NONE };
_t.text = mp; _t.tlen = strlen(mp);
return _t;
}
if (!more) {
Tok t = (Tok){ TK_EOF, start, "", 0, {0}, TK_NONE };
return t;
}
int c = lpeek(l, 0);
if (isidstart(c))
return lexident(l, start);
if (c >= '0' && c <= '9')
return lexnum(l, start);
if (c == '"') { lget(l); return lexstr(l, start); }
if (c == '\'') { lget(l); return lexrune(l, start); }
lget(l);
switch (c) {
case '(': EMIT(TK_LPAREN);
case ')': EMIT(TK_RPAREN);
case '{': EMIT(TK_LBRACE);
case '}': EMIT(TK_RBRACE);
case '[': EMIT(TK_LBRACK);
case ']': EMIT(TK_RBRACK);
case ',': EMIT(TK_COMMA);
case ';': EMIT(TK_SEMI);
case ':': EMIT(TK_COLON);
case '@': EMIT(TK_AT);
case '?': EMIT(TK_QUESTION);
case '~': EMIT(TK_TILDE);
case '.':
if (lpeek(l, 0) == '.' && lpeek(l, 1) == '.') {
lget(l); lget(l);
EMIT(TK_ELLIPSIS);
}
if (lpeek(l, 0) == '.') {
lget(l);
EMIT(TK_DOTDOT);
}
EMIT(TK_DOT);
case '+':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_PLUSEQ); }
EMIT(TK_PLUS);
case '-':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_MINUSEQ); }
if (lpeek(l, 0) == '>') { lget(l); EMIT(TK_ARROW); }
EMIT(TK_MINUS);
case '*':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_STAREQ); }
EMIT(TK_STAR);
case '/':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_SLASHEQ); }
EMIT(TK_SLASH);
case '%':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_PERCENTEQ); }
EMIT(TK_PERCENT);
case '&':
if (lpeek(l, 0) == '&') { lget(l); EMIT(TK_AND); }
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_AMPEQ); }
EMIT(TK_AMP);
case '|':
if (lpeek(l, 0) == '|') { lget(l); EMIT(TK_OR); }
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_PIPEEQ); }
EMIT(TK_PIPE);
case '^':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_CARETEQ); }
EMIT(TK_CARET);
case '=':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_EQ); }
if (lpeek(l, 0) == '>') { lget(l); EMIT(TK_FATARROW); }
EMIT(TK_ASSIGN);
case '!':
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_NEQ); }
EMIT(TK_NOT);
case '<':
if (lpeek(l, 0) == '<') {
lget(l);
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_LSHIFTEQ); }
EMIT(TK_LSHIFT);
}
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_LE); }
if (lpeek(l, 0) == '-') { lget(l); EMIT(TK_LARROW); }
EMIT(TK_LT);
case '>':
if (lpeek(l, 0) == '>') {
lget(l);
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_RSHIFTEQ); }
EMIT(TK_RSHIFT);
}
if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_GE); }
EMIT(TK_GT);
}
errorf(start, "unexpected character 0x%02x", c);
l->errs++;
Tok t = (Tok){ TK_ERR, start, NULL, 0, {0}, TK_NONE };
t.text = astrndup(l->a, (const char[]){ (char)c }, 1);
t.tlen = 1;
return t;
}