/* * No automatic semicolon insertion (Hare rule). The lexer only emits * what is in the source; the parser is responsible for non-empty rules. */ #include "ww.h" #include #include #include static int bomat(const char *src, u64 len, u64 pos) { return pos <= len && len - pos >= 3 && (unsigned char)src[pos] == 0xef && (unsigned char)src[pos + 1] == 0xbb && (unsigned char)src[pos + 2] == 0xbf; } void lexinit(Lex *l, Arena *a, const char *file, const char *src, u64 len) { memset(l, 0, sizeof *l); l->file = file; l->src = src; l->srclen = len; l->line = 1; l->col = 1; l->a = a; /* Go 1.26.5 syntax.source.nextch ignores U+FEFF only at the first * source position but counts its three bytes for the next column. */ if (bomat(src, len, 0)) { l->pos = 3; l->col = 4; } } static void lskipnul(Lex *l) { while (l->pos < l->srclen && l->src[l->pos] == '\0') { Pos p = { l->file, l->line, l->col }; l->pos++; l->col++; errorf(p, "invalid NUL character"); l->errs++; l->nulcount++; } } /* Return the raw offset of a logical byte lookahead. Raw NUL bytes do not * occupy a slot in the token stream: Go's source.nextch diagnoses them and * immediately resumes decoding at the following character. */ static u64 lrawoff(Lex *l, u64 ahead) { u64 p = l->pos; for (;;) { while (p < l->srclen && l->src[p] == '\0') p++; if (ahead == 0 || p >= l->srclen) return p; p++; ahead--; } } static int lpeek(Lex *l, u64 ahead) { /* Drain NUL at the current decoder position even when it is followed by * EOF; otherwise a trailing NUL could disappear without a diagnostic. */ for (;;) { if (l->pos >= l->srclen) return -1; int c = (unsigned char)l->src[l->pos]; if (c == 0) { lskipnul(l); continue; } if (ahead == 0) { if (bomat(l->src, l->srclen, l->pos)) return 0xfeff; return c; } u64 p = lrawoff(l, ahead); if (p >= l->srclen) return -1; if (bomat(l->src, l->srclen, p)) return 0xfeff; return (unsigned char)l->src[p]; } } static int lget(Lex *l) { for (;;) { if (l->pos >= l->srclen) return -1; int c = (unsigned char)l->src[l->pos]; if (c == 0) { lskipnul(l); continue; } if (bomat(l->src, l->srclen, l->pos)) { Pos p = { l->file, l->line, l->col }; l->pos += 3; l->col += 3; errorf(p, "invalid BOM in the middle of the file"); l->errs++; return 0xfeff; } l->pos++; if (c == '\n') { l->line++; l->col = 1; } else { l->col++; } return c; } } static Pos lpos(Lex *l) { Pos p = { l->file, l->line, l->col }; return p; } /* Copy one raw source span into token text while omitting diagnosed NUL * bytes. This keeps keyword, numeric, suffix, and directive recovery on the * same logical character stream as lpeek/lget. */ static char * lexspan(Lex *l, u64 begin, u64 end, u64 *len) { char *s = amalloc(l->a, end - begin + 1); u64 j = 0; for (u64 i = begin; i < end; i++) if (l->src[i] != '\0') s[j++] = l->src[i]; s[j] = '\0'; *len = j; return s; } static int isidstart(int c) { return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || c == '_'; } static int isidcont(int c) { return isidstart(c) || (c >= '0' && c <= '9'); } static int ishex(int c) { return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'); } /* Consume one // body, then classify the driver's internal module directive * from its logical (NUL-filtered) text. Keeping a single moving lexer cursor * makes even long generated module paths linear rather than repeatedly * rescanning from the start of the comment. The trailing newline remains for * the ordinary whitespace loop. */ static void linecomment(Lex *l) { u64 begin = l->pos; u64 nulbegin = l->nulcount; int c; while ((c = lpeek(l, 0)) >= 0 && c != '\n') lget(l); u64 end = l->pos; u64 n; const char *body; if (l->nulcount == nulbegin) { n = end - begin; body = l->src + begin; } else { body = lexspan(l, begin, end, &n); } static const char pre[] = "ww:module"; const u64 plen = sizeof pre - 1; if (n < plen || memcmp(body, pre, plen) != 0 || n == plen) return; u64 i = plen; if (body[i] == '-') { static const char rest[] = "-reset"; const u64 rlen = sizeof rest - 1; if (n - i < rlen || memcmp(body + i, rest, rlen) != 0) return; i += rlen; if (i == n) { l->modreset = 1; l->modpath = NULL; /* #9: reset supersedes a pending path */ return; } if (body[i] != ' ' && body[i] != '\t') return; while (i < n && (body[i] == ' ' || body[i] == '\t')) i++; u64 s = i; while (i < n && body[i] != '\r' && body[i] != ' ' && body[i] != '\t') i++; l->modreset = 1; l->modpath = NULL; /* #9: see above */ if (i > s) l->modresetpath = astrndup(l->a, body + s, i - s); return; } if (body[i] != ' ' && body[i] != '\t') return; while (i < n && (body[i] == ' ' || body[i] == '\t')) i++; u64 s = i; while (i < n && body[i] != '\r' && body[i] != ' ' && body[i] != '\t') i++; if (i > s) l->modpath = astrndup(l->a, body + s, i - s); } static int skipws(Lex *l) { for (;;) { int c = lpeek(l, 0); if (c < 0) return 0; if (c == ' ' || c == '\t' || c == '\r' || c == '\n') { lget(l); continue; } if (c == '/' && lpeek(l, 1) == '/') { lget(l); lget(l); /* #16 option-B: classify the driver's whole-line internal * module boundary after consuming it once. */ linecomment(l); continue; } if (c == '/' && lpeek(l, 1) == '*') { lget(l); lget(l); int prev = -1; for (;;) { int x = lget(l); if (x < 0) { Pos p = lpos(l); errorf(p, "unterminated /* comment"); l->errs++; return 0; } if (prev == '*' && x == '/') break; prev = x; } continue; } return 1; } } static u64 parseint(const char *s, u64 n, int base, int *ok) { u64 v = 0; u64 b = (u64)base; u64 cutoff = (u64)~0ULL / b; u64 cutlim = (u64)~0ULL % b; int got = 0; for (u64 i = 0; i < n; i++) { int c = (unsigned char)s[i]; if (c == '_') continue; int d; if (c >= '0' && c <= '9') d = c - '0'; else if (c >= 'a' && c <= 'f') d = c - 'a' + 10; else if (c >= 'A' && c <= 'F') d = c - 'A' + 10; else { *ok = 0; return 0; } if (d >= base) { *ok = 0; return 0; } if (v > cutoff || (v == cutoff && (u64)d > cutlim)) { *ok = 0; return 0; } v = v * b + (u64)d; got = 1; } *ok = got; return v; } /* shared escape decoder for \xHH (n=2), \uHHHH (n=4), \UHHHHHHHH (n=8). * All three yield a codepoint, not a raw byte — mirrors * ref/hare/hare/lex/lex.ha:347 fn lex_unicode. Error strings copied * verbatim from that reference for diagnostic fidelity (#50). */ static int lexunicode(Lex *l, int n, int *out) { u32 u = 0; for (int i = 0; i < n; i++) { int c = lget(l); if (c < 0) { Pos p = lpos(l); errorf(p, "unexpected EOF scanning for escape"); l->errs++; return -1; } if (!ishex(c)) { Pos p = lpos(l); errorf(p, "unexpected rune scanning for escape"); l->errs++; return -1; } int d = (c <= '9' ? c - '0' : (c | 0x20) - 'a' + 10); u = (u << 4) | (u32)d; } if (u > 0x10FFFF || (u >= 0xD800 && u < 0xE000)) { Pos p = lpos(l); errorf(p, "invalid unicode codepoint in escape"); l->errs++; return -1; } *out = (int)u; return 0; } /* utf8enc — encode codepoint cp (already validated <= 0x10FFFF and * non-surrogate by lexunicode) into out (>= 4 bytes), return the byte * count. The C bootstrap has no stdlib; this mirrors lib/encoding/utf8 * encoderune (the ww side calls that directly). */ static int utf8enc(u32 cp, char *out) { if (cp < 0x80) { out[0] = (char)cp; return 1; } else if (cp < 0x800) { out[0] = (char)(0xC0 | (cp >> 6)); out[1] = (char)(0x80 | (cp & 0x3F)); return 2; } else if (cp < 0x10000) { out[0] = (char)(0xE0 | (cp >> 12)); out[1] = (char)(0x80 | ((cp >> 6) & 0x3F)); out[2] = (char)(0x80 | (cp & 0x3F)); return 3; } out[0] = (char)(0xF0 | (cp >> 18)); out[1] = (char)(0x80 | ((cp >> 12) & 0x3F)); out[2] = (char)(0x80 | ((cp >> 6) & 0x3F)); out[3] = (char)(0x80 | (cp & 0x3F)); return 4; } static int escape(Lex *l, int *out) { int c = lget(l); if (c < 0) return -1; switch (c) { case 'n': *out = '\n'; return 0; case 't': *out = '\t'; return 0; case 'r': *out = '\r'; return 0; case '\\': *out = '\\'; return 0; case '\'': *out = '\''; return 0; case '"': *out = '"'; return 0; case '0': *out = '\0'; return 0; case 'a': *out = '\a'; return 0; case 'b': *out = '\b'; return 0; case 'f': *out = '\f'; return 0; case 'v': *out = '\v'; return 0; case 'x': return lexunicode(l, 2, out); case 'u': return lexunicode(l, 4, out); case 'U': return lexunicode(l, 8, out); } { Pos p = lpos(l); errorf(p, "bad escape \\%c", c); l->errs++; } return -1; } static const char * lextypesuffix(Lex *l, u64 *len) { char got[4]; u64 n = 0; while (n < sizeof got && isidcont(lpeek(l, n))) { got[n] = (char)lpeek(l, n); n++; } if (n == 0 || n > 3 || isidcont(lpeek(l, n))) return NULL; static const char *const names[] = { "i8", "i16", "i32", "i64", "u8", "u16", "u32", "u64", "f32", "f64", NULL }; for (int i = 0; names[i]; i++) { u64 nl = strlen(names[i]); if (nl == n && memcmp(names[i], got, n) == 0) { *len = n; return names[i]; } } return NULL; } static Tok lexnum(Lex *l, Pos start) { Tok t = (Tok){ TK_INT, start, NULL, 0, {0}, TK_NONE }; u64 begin = l->pos; u64 nulbegin = l->nulcount; int base = 10; int isfloat = 0; int c = lpeek(l, 0); if (c == '0' && (lpeek(l, 1) == 'x' || lpeek(l, 1) == 'X')) { lget(l); lget(l); base = 16; while ((c = lpeek(l, 0)) >= 0 && (ishex(c) || c == '_')) lget(l); } else if (c == '0' && (lpeek(l, 1) == 'b' || lpeek(l, 1) == 'B')) { lget(l); lget(l); base = 2; while ((c = lpeek(l, 0)) >= 0 && (c == '0' || c == '1' || c == '_')) lget(l); } else if (c == '0' && (lpeek(l, 1) == 'o' || lpeek(l, 1) == 'O')) { lget(l); lget(l); base = 8; while ((c = lpeek(l, 0)) >= 0 && ((c >= '0' && c <= '7') || c == '_')) lget(l); } else { while ((c = lpeek(l, 0)) >= 0 && ((c >= '0' && c <= '9') || c == '_')) lget(l); if (lpeek(l, 0) == '.' && lpeek(l, 1) >= '0' && lpeek(l, 1) <= '9') { isfloat = 1; lget(l); while ((c = lpeek(l, 0)) >= 0 && ((c >= '0' && c <= '9') || c == '_')) lget(l); c = lpeek(l, 0); if (c == 'e' || c == 'E') { lget(l); if (lpeek(l, 0) == '+' || lpeek(l, 0) == '-') lget(l); while ((c = lpeek(l, 0)) >= 0 && c >= '0' && c <= '9') lget(l); } } } u64 rawend = l->pos; u64 n; if (l->nulcount == nulbegin) { n = rawend - begin; t.text = astrndup(l->a, l->src + begin, n); } else { t.text = lexspan(l, begin, rawend, &n); } t.tlen = n; if (isfloat) { t.kind = TK_FLOAT; /* strdup with underscores stripped before strtod */ char *clean = amalloc(l->a, n + 1); u64 j = 0; for (u64 i = 0; i < n; i++) if (t.text[i] != '_') clean[j++] = t.text[i]; clean[j] = '\0'; errno = 0; t.v.fval = strtod(clean, NULL); if (errno) { errorf(start, "bad float literal '%s'", t.text); l->errs++; } } else { const char *digs = t.text; u64 dn = n; if (base != 10) { digs += 2; dn -= 2; } int ok = 0; t.v.uval = parseint(digs, dn, base, &ok); if (!ok) { errorf(start, "bad integer literal '%s'", t.text); l->errs++; t.kind = TK_ERR; } } /* A typed suffix must be glued (no whitespace) to the digits. */ if (isidstart(lpeek(l, 0))) { u64 sl; const char *match = lextypesuffix(l, &sl); if (match) { for (u64 i = 0; i < sl; i++) lget(l); t.tsuffix = astrndup(l->a, match, sl); } } return t; } static Tok lexident(Lex *l, Pos start) { u64 begin = l->pos; u64 nulbegin = l->nulcount; while (isidcont(lpeek(l, 0))) lget(l); u64 n; const char *p; int filtered = l->nulcount != nulbegin; if (!filtered) { n = l->pos - begin; p = l->src + begin; } else { p = lexspan(l, begin, l->pos, &n); } char *text = filtered ? (char *)p : astrndup(l->a, p, n); /* bare '_' is the discard marker. `_x`, `_1` are normal idents. */ if (n == 1 && p[0] == '_') { Tok t = (Tok){ TK_UNDER, start, text, n, {0}, TK_NONE }; return t; } Tkind k = kwlookup(p, n); Tok t = (Tok){ k != TK_NONE ? k : TK_IDENT, start, text, n, {0}, TK_NONE }; return t; } static Tok lexstr(Lex *l, Pos start) { /* opening quote already consumed by caller */ u64 cap = 32, n = 0; char *buf = amalloc(l->a, cap); for (;;) { int c = lpeek(l, 0); if (c < 0) { errorf(start, "unterminated string"); l->errs++; Tok t = (Tok){ TK_ERR, start, astrndup(l->a, "", 0), 0, {0}, TK_NONE }; return t; } if (c == '"') { lget(l); break; } /* Escape-decoded values are codepoints and UTF-8-encode into * 1-4 bytes (mirrors Hare's memio::appendrune in lex_string, * ref/hare/hare/lex/lex.ha:431). Raw source bytes are already * UTF-8 and pass through unchanged — re-encoding them would * double-encode the >0x7F continuation bytes. */ char enc[4]; int el; if (c == '\\') { int ch; lget(l); if (escape(l, &ch) < 0) ch = 0; el = utf8enc((u32)ch, enc); } else { enc[0] = (char)lget(l); el = 1; } if (n + el >= cap) { u64 ncap = cap * 2; while (n + el >= ncap) ncap *= 2; char *nb = amalloc(l->a, ncap); memcpy(nb, buf, n); buf = nb; cap = ncap; } for (int i = 0; i < el; i++) buf[n++] = enc[i]; } buf[n] = '\0'; Tok t = (Tok){ TK_STR, start, buf, n, {0}, TK_NONE }; return t; } static Tok lexrune(Lex *l, Pos start) { int ch; int c = lpeek(l, 0); if (c < 0) { errorf(start, "unterminated rune"); l->errs++; return (Tok){ TK_ERR, start, "", 0, {0}, TK_NONE }; } if (c == '\\') { lget(l); if (escape(l, &ch) < 0) ch = 0; } else { ch = lget(l); } if (lpeek(l, 0) != '\'') { errorf(start, "rune literal missing closing '"); l->errs++; return (Tok){ TK_ERR, start, "", 0, {0}, TK_NONE }; } lget(l); Tok t = (Tok){ TK_RUNE, start, NULL, 0, {0}, TK_NONE }; t.v.uval = (u64)(u32)ch; t.text = aprintf(l->a, "%d", ch); t.tlen = strlen(t.text); return t; } #define EMIT(K) do { Tok _t = (Tok){ (K), start, NULL, 0, {0}, TK_NONE }; \ _t.text = tokname(K); _t.tlen = strlen(_t.text); return _t; } while (0) Tok lexnext(Lex *l) { int more = skipws(l); Pos start = lpos(l); /* A `//ww:module-reset` seen in the skipped run surfaces as its own * token before the next real one (#16 option-B boundary reset). */ if (l->modreset) { l->modreset = 0; const char *rp = l->modresetpath; l->modresetpath = NULL; /* path-carrying reset → text=path (#57); bare reset → text=NULL */ Tok _t = (Tok){ TK_MODRESET, start, rp, rp ? strlen(rp) : 0, {0}, TK_NONE }; return _t; } if (l->modpath) { const char *mp = l->modpath; l->modpath = NULL; Tok _t = (Tok){ TK_MODPATH, start, NULL, 0, {0}, TK_NONE }; _t.text = mp; _t.tlen = strlen(mp); return _t; } if (!more) { Tok t = (Tok){ TK_EOF, start, "", 0, {0}, TK_NONE }; return t; } int c = lpeek(l, 0); if (c == 0xfeff) { lget(l); Tok t = (Tok){ TK_ERR, start, "", 0, {0}, TK_NONE }; return t; } if (isidstart(c)) return lexident(l, start); if (c >= '0' && c <= '9') return lexnum(l, start); if (c == '"') { lget(l); return lexstr(l, start); } if (c == '\'') { lget(l); return lexrune(l, start); } lget(l); switch (c) { case '(': EMIT(TK_LPAREN); case ')': EMIT(TK_RPAREN); case '{': EMIT(TK_LBRACE); case '}': EMIT(TK_RBRACE); case '[': EMIT(TK_LBRACK); case ']': EMIT(TK_RBRACK); case ',': EMIT(TK_COMMA); case ';': EMIT(TK_SEMI); case ':': EMIT(TK_COLON); case '@': EMIT(TK_AT); case '?': EMIT(TK_QUESTION); case '~': EMIT(TK_TILDE); case '.': if (lpeek(l, 0) == '.' && lpeek(l, 1) == '.') { lget(l); lget(l); EMIT(TK_ELLIPSIS); } if (lpeek(l, 0) == '.') { lget(l); EMIT(TK_DOTDOT); } EMIT(TK_DOT); case '+': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_PLUSEQ); } EMIT(TK_PLUS); case '-': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_MINUSEQ); } if (lpeek(l, 0) == '>') { lget(l); EMIT(TK_ARROW); } EMIT(TK_MINUS); case '*': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_STAREQ); } EMIT(TK_STAR); case '/': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_SLASHEQ); } EMIT(TK_SLASH); case '%': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_PERCENTEQ); } EMIT(TK_PERCENT); case '&': if (lpeek(l, 0) == '&') { lget(l); EMIT(TK_AND); } if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_AMPEQ); } EMIT(TK_AMP); case '|': if (lpeek(l, 0) == '|') { lget(l); EMIT(TK_OR); } if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_PIPEEQ); } EMIT(TK_PIPE); case '^': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_CARETEQ); } EMIT(TK_CARET); case '=': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_EQ); } if (lpeek(l, 0) == '>') { lget(l); EMIT(TK_FATARROW); } EMIT(TK_ASSIGN); case '!': if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_NEQ); } EMIT(TK_NOT); case '<': if (lpeek(l, 0) == '<') { lget(l); if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_LSHIFTEQ); } EMIT(TK_LSHIFT); } if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_LE); } if (lpeek(l, 0) == '-') { lget(l); EMIT(TK_LARROW); } EMIT(TK_LT); case '>': if (lpeek(l, 0) == '>') { lget(l); if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_RSHIFTEQ); } EMIT(TK_RSHIFT); } if (lpeek(l, 0) == '=') { lget(l); EMIT(TK_GE); } EMIT(TK_GT); } errorf(start, "unexpected character 0x%02x", c); l->errs++; Tok t = (Tok){ TK_ERR, start, NULL, 0, {0}, TK_NONE }; t.text = astrndup(l->a, (const char[]){ (char)c }, 1); t.tlen = 1; return t; }