diff --git a/cmd/wcc/lex.c b/cmd/wcc/lex.c index 1a89c5a0..31c5acb5 100644 --- a/cmd/wcc/lex.c +++ b/cmd/wcc/lex.c @@ -34,38 +34,91 @@ lexinit(Lex *l, Arena *a, const char *file, const char *src, u64 len) } } +static void +lskipnul(Lex *l) +{ + while (l->pos < l->srclen && l->src[l->pos] == '\0') { + Pos p = { l->file, l->line, l->col }; + l->pos++; + l->col++; + errorf(p, "invalid NUL character"); + l->errs++; + l->nulcount++; + } +} + +/* Return the raw offset of a logical byte lookahead. Raw NUL bytes do not + * occupy a slot in the token stream: Go's source.nextch diagnoses them and + * immediately resumes decoding at the following character. */ +static u64 +lrawoff(Lex *l, u64 ahead) +{ + u64 p = l->pos; + for (;;) { + while (p < l->srclen && l->src[p] == '\0') + p++; + if (ahead == 0 || p >= l->srclen) + return p; + p++; + ahead--; + } +} + static int lpeek(Lex *l, u64 ahead) { - u64 p = l->pos + ahead; - if (p >= l->srclen) - return -1; - if (bomat(l->src, l->srclen, p)) - return 0xfeff; - return (unsigned char)l->src[p]; + /* Drain NUL at the current decoder position even when it is followed by + * EOF; otherwise a trailing NUL could disappear without a diagnostic. */ + for (;;) { + if (l->pos >= l->srclen) + return -1; + int c = (unsigned char)l->src[l->pos]; + if (c == 0) { + lskipnul(l); + continue; + } + if (ahead == 0) { + if (bomat(l->src, l->srclen, l->pos)) + return 0xfeff; + return c; + } + u64 p = lrawoff(l, ahead); + if (p >= l->srclen) + return -1; + if (bomat(l->src, l->srclen, p)) + return 0xfeff; + return (unsigned char)l->src[p]; + } } static int lget(Lex *l) { - if (l->pos >= l->srclen) - return -1; - if (bomat(l->src, l->srclen, l->pos)) { - Pos p = { l->file, l->line, l->col }; - l->pos += 3; - l->col += 3; - errorf(p, "invalid BOM in the middle of the file"); - l->errs++; - return 0xfeff; + for (;;) { + if (l->pos >= l->srclen) + return -1; + int c = (unsigned char)l->src[l->pos]; + if (c == 0) { + lskipnul(l); + continue; + } + if (bomat(l->src, l->srclen, l->pos)) { + Pos p = { l->file, l->line, l->col }; + l->pos += 3; + l->col += 3; + errorf(p, "invalid BOM in the middle of the file"); + l->errs++; + return 0xfeff; + } + l->pos++; + if (c == '\n') { + l->line++; + l->col = 1; + } else { + l->col++; + } + return c; } - int c = (unsigned char)l->src[l->pos++]; - if (c == '\n') { - l->line++; - l->col = 1; - } else { - l->col++; - } - return c; } static Pos @@ -75,6 +128,22 @@ lpos(Lex *l) return p; } +/* Copy one raw source span into token text while omitting diagnosed NUL + * bytes. This keeps keyword, numeric, suffix, and directive recovery on the + * same logical character stream as lpeek/lget. */ +static char * +lexspan(Lex *l, u64 begin, u64 end, u64 *len) +{ + char *s = amalloc(l->a, end - begin + 1); + u64 j = 0; + for (u64 i = begin; i < end; i++) + if (l->src[i] != '\0') + s[j++] = l->src[i]; + s[j] = '\0'; + *len = j; + return s; +} + static int isidstart(int c) { @@ -94,6 +163,70 @@ ishex(int c) (c >= 'A' && c <= 'F'); } +/* Consume one // body, then classify the driver's internal module directive + * from its logical (NUL-filtered) text. Keeping a single moving lexer cursor + * makes even long generated module paths linear rather than repeatedly + * rescanning from the start of the comment. The trailing newline remains for + * the ordinary whitespace loop. */ +static void +linecomment(Lex *l) +{ + u64 begin = l->pos; + u64 nulbegin = l->nulcount; + int c; + while ((c = lpeek(l, 0)) >= 0 && c != '\n') + lget(l); + u64 end = l->pos; + u64 n; + const char *body; + if (l->nulcount == nulbegin) { + n = end - begin; + body = l->src + begin; + } else { + body = lexspan(l, begin, end, &n); + } + + static const char pre[] = "ww:module"; + const u64 plen = sizeof pre - 1; + if (n < plen || memcmp(body, pre, plen) != 0 || n == plen) + return; + u64 i = plen; + if (body[i] == '-') { + static const char rest[] = "-reset"; + const u64 rlen = sizeof rest - 1; + if (n - i < rlen || memcmp(body + i, rest, rlen) != 0) + return; + i += rlen; + if (i == n) { + l->modreset = 1; + l->modpath = NULL; /* #9: reset supersedes a pending path */ + return; + } + if (body[i] != ' ' && body[i] != '\t') + return; + while (i < n && (body[i] == ' ' || body[i] == '\t')) + i++; + u64 s = i; + while (i < n && body[i] != '\r' && body[i] != ' ' + && body[i] != '\t') + i++; + l->modreset = 1; + l->modpath = NULL; /* #9: see above */ + if (i > s) + l->modresetpath = astrndup(l->a, body + s, i - s); + return; + } + if (body[i] != ' ' && body[i] != '\t') + return; + while (i < n && (body[i] == ' ' || body[i] == '\t')) + i++; + u64 s = i; + while (i < n && body[i] != '\r' && body[i] != ' ' && body[i] != '\t') + i++; + if (i > s) + l->modpath = astrndup(l->a, body + s, i - s); +} + static int skipws(Lex *l) { @@ -107,78 +240,9 @@ skipws(Lex *l) } if (c == '/' && lpeek(l, 1) == '/') { lget(l); lget(l); - /* #16 option-B: the driver emits `//ww:module-reset` - * before a package-less file's bytes; recognize the - * whole-line directive (without consuming differently) - * and flag it so lexnext emits TK_MODRESET. The body is - * then skipped like any comment. Mirrors the removed - * `// MODULE:` lexer directive. */ - { - static const char pre[] = "ww:module"; - size_t i = 0; - while (pre[i] && lpeek(l, i) == pre[i]) - i++; - if (pre[i] == '\0') { - int nx = lpeek(l, i); - if (nx == '-') { - static const char rest[] = "-reset"; - size_t j = 0; - while (rest[j] && lpeek(l, i + j) == rest[j]) - j++; - if (rest[j] == '\0') { - int af = lpeek(l, i + j); - if (af == '\n' || af < 0) { - l->modreset = 1; - l->modpath = NULL; /* #9: a reset supersedes a path opened - * earlier in this skipws run (empty/ - * export-less inlined module body) */ - } else if (af == ' ' - || af == '\t') { - /* `//ww:module-reset ` - * — sep primary body tagged by - * its full dotted import path so - * definer == importer (#57). */ - size_t k = i + j; - while (lpeek(l, k) == ' ' - || lpeek(l, k) == '\t') - k++; - size_t s = k; - int dch; - while ((dch = lpeek(l, k)) >= 0 - && dch != '\n' - && dch != '\r' - && dch != ' ' - && dch != '\t') - k++; - l->modreset = 1; - l->modpath = NULL; /* #9: see above — clear pending path */ - if (k > s) - l->modresetpath = - astrndup(l->a, - l->src + l->pos + s, - k - s); - } - } - } else if (nx == ' ' || nx == '\t') { - /* `//ww:module ` — M1 import boundary. */ - size_t k = i; - while (lpeek(l, k) == ' ' - || lpeek(l, k) == '\t') - k++; - size_t s = k; - int ch; - while ((ch = lpeek(l, k)) >= 0 - && ch != '\n' && ch != '\r' - && ch != ' ' && ch != '\t') - k++; - if (k > s) - l->modpath = astrndup(l->a, - l->src + l->pos + s, k - s); - } - } - } - while ((c = lpeek(l, 0)) >= 0 && c != '\n') - lget(l); + /* #16 option-B: classify the driver's whole-line internal + * module boundary after consuming it once. */ + linecomment(l); continue; } if (c == '/' && lpeek(l, 1) == '*') { @@ -318,11 +382,38 @@ escape(Lex *l, int *out) return -1; } +static const char * +lextypesuffix(Lex *l, u64 *len) +{ + char got[4]; + u64 n = 0; + while (n < sizeof got && isidcont(lpeek(l, n))) { + got[n] = (char)lpeek(l, n); + n++; + } + if (n == 0 || n > 3 || isidcont(lpeek(l, n))) + return NULL; + static const char *const names[] = { + "i8", "i16", "i32", "i64", + "u8", "u16", "u32", "u64", + "f32", "f64", NULL + }; + for (int i = 0; names[i]; i++) { + u64 nl = strlen(names[i]); + if (nl == n && memcmp(names[i], got, n) == 0) { + *len = n; + return names[i]; + } + } + return NULL; +} + static Tok lexnum(Lex *l, Pos start) { Tok t = (Tok){ TK_INT, start, NULL, 0, {0}, TK_NONE }; u64 begin = l->pos; + u64 nulbegin = l->nulcount; int base = 10; int isfloat = 0; int c = lpeek(l, 0); @@ -361,8 +452,14 @@ lexnum(Lex *l, Pos start) } } - u64 n = l->pos - begin; - t.text = astrndup(l->a, l->src + begin, n); + u64 rawend = l->pos; + u64 n; + if (l->nulcount == nulbegin) { + n = rawend - begin; + t.text = astrndup(l->a, l->src + begin, n); + } else { + t.text = lexspan(l, begin, rawend, &n); + } t.tlen = n; if (isfloat) { @@ -371,8 +468,8 @@ lexnum(Lex *l, Pos start) char *clean = amalloc(l->a, n + 1); u64 j = 0; for (u64 i = 0; i < n; i++) - if (l->src[begin + i] != '_') - clean[j++] = l->src[begin + i]; + if (t.text[i] != '_') + clean[j++] = t.text[i]; clean[j] = '\0'; errno = 0; t.v.fval = strtod(clean, NULL); @@ -381,7 +478,7 @@ lexnum(Lex *l, Pos start) l->errs++; } } else { - const char *digs = l->src + begin; + const char *digs = t.text; u64 dn = n; if (base != 10) { digs += 2; @@ -398,28 +495,12 @@ lexnum(Lex *l, Pos start) /* A typed suffix must be glued (no whitespace) to the digits. */ if (isidstart(lpeek(l, 0))) { - u64 sb = l->pos; - i32 sc = l->col; - while (isidcont(lpeek(l, 0))) lget(l); - u64 sl = l->pos - sb; - const char *names[] = { - "i8", "i16", "i32", "i64", - "u8", "u16", "u32", "u64", - "f32", "f64", NULL - }; - const char *match = NULL; - for (int i = 0; names[i]; i++) { - u64 nl = strlen(names[i]); - if (nl == sl && memcmp(names[i], l->src + sb, nl) == 0) { - match = names[i]; - break; - } - } + u64 sl; + const char *match = lextypesuffix(l, &sl); if (match) { - t.tsuffix = astrndup(l->a, l->src + sb, sl); - } else { - l->pos = sb; - l->col = sc; + for (u64 i = 0; i < sl; i++) + lget(l); + t.tsuffix = astrndup(l->a, match, sl); } } return t; @@ -429,18 +510,27 @@ static Tok lexident(Lex *l, Pos start) { u64 begin = l->pos; + u64 nulbegin = l->nulcount; while (isidcont(lpeek(l, 0))) lget(l); - u64 n = l->pos - begin; - const char *p = l->src + begin; + u64 n; + const char *p; + int filtered = l->nulcount != nulbegin; + if (!filtered) { + n = l->pos - begin; + p = l->src + begin; + } else { + p = lexspan(l, begin, l->pos, &n); + } + char *text = filtered ? (char *)p : astrndup(l->a, p, n); /* bare '_' is the discard marker. `_x`, `_1` are normal idents. */ if (n == 1 && p[0] == '_') { - Tok t = (Tok){ TK_UNDER, start, astrndup(l->a, p, n), n, {0}, TK_NONE }; + Tok t = (Tok){ TK_UNDER, start, text, n, {0}, TK_NONE }; return t; } Tkind k = kwlookup(p, n); Tok t = (Tok){ k != TK_NONE ? k : TK_IDENT, start, - astrndup(l->a, p, n), n, {0}, TK_NONE }; + text, n, {0}, TK_NONE }; return t; } diff --git a/cmd/wcc/ww.h b/cmd/wcc/ww.h index 0fd38ec3..af63ea68 100644 --- a/cmd/wcc/ww.h +++ b/cmd/wcc/ww.h @@ -202,6 +202,7 @@ struct Lex { i32 col; Arena *a; /* token-text arena */ int errs; + u64 nulcount; /* raw NUL bytes diagnosed by the source decoder */ int modreset; /* a `//ww:module-reset` directive was seen in * the last skipped run; lexnext emits TK_MODRESET * before the next real token. */ diff --git a/docs/build-system.md b/docs/build-system.md index dbd43d02..a7012d31 100644 --- a/docs/build-system.md +++ b/docs/build-system.md @@ -8545,6 +8545,119 @@ and diagnostic text is not a persisted-byte contract. Build workdir format remains `18`, test workdir format remains `19`, and semantic storage format remains `3`. +### 11.43 Implemented selected-source U+0000 rejection + +Every selected physical `.ww` source rejects a raw byte `00` (U+0000) at its +physical line and byte column with exactly `invalid NUL character`. The rule is +source-wide: comments, interpreted-string text, rune text, and between-token +positions cannot turn a raw NUL into payload. An escape spelling such as +`\x00` remains a legal literal value because it is not byte `00` in +the source file. This section is only the raw-U+0000 rule: malformed UTF-8 +and the independently implemented per-source BOM boundary in §11.42 are not +changed or broadened here. + +#### Pinned evidence, applicability, and measured prior behavior + +The sole authority is official Go 1.26.5 at +`c19862e5f8415b4f24b189d065ed739517c548ba`. Its compiler owner, +`(*source).nextch` in +[`cmd/compile/internal/syntax/source.go`, lines 113–165](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/cmd/compile/internal/syntax/source.go#L113-L165), +detects ASCII zero at lines 121–129, reports `invalid NUL character`, and +continues decoding. `TestScanErrors` pins the positioned diagnostic at +[`cmd/compile/internal/syntax/scanner_test.go`, lines 587–600](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/cmd/compile/internal/syntax/scanner_test.go#L587-L600), +and compiler testdata [`test/nul1.go`, lines 7–52](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/test/nul1.go#L7-L52) +requires NUL errors in strings, raw strings, line/block comments, and ordinary +source. Go's independent public scanner has the same rule in +[`go/scanner/scanner.go`, lines 63–108](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/go/scanner/scanner.go#L63-L108) +and its tests at lines 790–819. These are **behavior directly implemented or +asserted by pinned Go**. + +Before this change, directly measured Cstage and WWstage package builds both +accepted a NUL in a line comment, wrote it into the committed synthetic unit, +published byte-identical archive/interface products, and allowed a dotted local +importer to link and run. The retained audit commands used `out/bin/ww` and +`out/bin/ww_ww` with `WW_SRCLIB=/home/kimchi/src/ww/lib` over +`/tmp/ww-pkgaudit.qtjcpO/src/nulcomment`; both exited zero with empty streams. +The same acceptance applied to raw NUL in strings and runes. Those observations +are **directly measured WW behavior**. Applying pinned Go's physical-source +rule independently to WW's selected files is **behavior derived from the +pinned implementation** and is applicable without importing Go's module, +manifest, registry, lock, cache, database, CAS, network, generalized-import, +or source-level build-expression model. + +#### Ownership, timing, and four axes + +The direct-frontend semantic owners are the logical source-decoder helpers in +`cmd/wcc/lex.c` and their exact twins in `lib/ww/syntax/lex.ww`: each raw zero +is consumed, reports the same positioned error, and is omitted before token +recovery. Filtered token spans preserve that decoder rule through identifiers, +numbers and suffixes, directives, escapes, operators, comments, and EOF. The +line-comment decoder consumes each body once before classifying an internal +module directive, so NUL filtering does not make generated path handling +superlinear. The +shared `internal/wwpackage/package.ww` coordinator owns public package/test +diagnostic precedence: its length-aware preflight reports every raw NUL in an +invalid physical source before its manual package-clause classifier. It is not +another identity policy. Driver source slurping and synthetic-unit composition +preserve byte lengths and are not semantic owners. + +- **Go-like build:** after fixed-target filename selection, an invalid selected + root or dependency rejects during loading, before graph completion and before + compiler, assembler, archiver, linker, install, publication, or execution. + Source rejection in a root precedes resolution of that root's missing + imports. A wrong-target file is excluded before this rule and remains unread + by its semantic owners. +- **Go-like test:** selected production, same-package, external-package, and + test-only source each receive the rule before variant construction. Failure + emits the established attributable `FAIL\n` without a generated main, test + process, accounting, `ok` result, retained binary, or public test product. +- **Go-like package:** each selected physical file owns its diagnostic and + position. Declared package name, source role, package conflict handling, + command/test family, physical directory, and exact canonical dotted identity + are unchanged. +- **Go-like import:** invalid bytes create no import edge or graph node. Valid + import spelling, aliases, local/vendor/internal resolution, visibility, + cycle handling, and initialization order remain unchanged. + +Thus raw NUL is never an input to manifest-free package identity, graph/action +keys, symbols, `.wwi`, archive naming, publication names, or persistence keys. +It neither changes local dotted-import boundaries nor introduces a manifest. + +#### Failure, publication, persistence, and parity + +Cold invalid requests create no unit, assembly, object, archive, executable, +capture, `.new`, `.install`, `.wwtxn.*`, or public output. A warm edit that +introduces NUL stops before a producer or install action, preserving the prior +committed unit/interface/assembly/object/archive generation, tool vouchers, +stamp, and public output byte for byte. Removing the NUL restores the ordinary +selected-source fingerprint; exact restoration may reuse the earlier generation. +Existing producer/runtime failure and transaction rollback remain their own +owners because this branch creates no new rollback mechanism. + +Lexer and coordinator state are request/source-local. Concurrent valid and +invalid requests keep independent workdirs, captures, diagnostics, and +products; an invalid request cannot contaminate a valid sibling. The rule adds +no process, wait, or cancellation boundary, so signal, timeout, interruption, +process-group cleanup, and ordinary scratch cleanup retain their established +owners. Validation itself leaves no durable residue. Cstage and WWstage are +semantic twins: diagnostics match exactly, and valid unit/compiler/product +bytes retain their existing byte-identity contract. + +`raw_nul_is_rejected_in_every_selected_source` in +`test/package/package_test.ww`, with focused C and WW lexer coverage, proves +literal and comment contexts plus adjacency recovery across escapes, +identifiers, numbers and typed suffixes, operators, comment delimiters, and +EOF; it also proves selected/imported builds, wrong-target exclusion, every +directory-test source role, precedence, cold cleanup, warm rollback and reuse, +concurrent isolation, valid escaped-NUL behavior, runtime/publication behavior, +and no-residue/parity observations. + +No format bump. This changes invalid-source acceptance and diagnostics only; +valid source composition and valid `.wwi`, assembly, object, archive, +executable, and retained-test-product bytes are unchanged. Build workdir format +remains `18`, test workdir format remains `19`, and semantic storage format +remains `3`. + ## 12. Candidate architectures and hard-gate decision Five candidates were developed as coherent systems, not as feature bins. diff --git a/docs/spec.md b/docs/spec.md index 60eaf3c4..e962488e 100644 --- a/docs/spec.md +++ b/docs/spec.md @@ -60,6 +60,16 @@ every other source position, including inside string and rune literals and comments. Apart from that marker rule, the lexer operates on bytes and non-ASCII bytes are legal only inside string and rune literals and comments. +Each raw byte `00` (U+0000) is invalid at every physical source position, +including in comments and string or rune literal text. It produces one +positioned `invalid NUL character` error at that byte's source position and +is omitted from the lexer's logical character stream before token recovery. +It therefore cannot split an identifier, number, operator, escape, or comment +boundary into different tokens. This is a source-representation rule, before +package/import interpretation; it does not make U+0000 a package, import, or +artifact identity component. An escape spelling such as `"\\x00"` is not a +raw source byte and remains a legal literal value. + ### 2.2 Comments Line comments only, introduced by `//` and running to end of line. There diff --git a/docs/test-system-v2.md b/docs/test-system-v2.md index 08addab2..53613e20 100644 --- a/docs/test-system-v2.md +++ b/docs/test-system-v2.md @@ -429,6 +429,26 @@ before either operation. The marker is source representation only and never package/import/action/artifact/publication/persistence identity; no test result is cached. +After filename eligibility and before package-clause classification, each raw +source byte `00` in a selected production, same-package test, +external-package test, or test-only source is one positioned +`invalid NUL character` error. Comments and literal text do not hide it. The +shared coordinator owns that early package/test rejection and reports every +raw NUL in the invalid physical source. On complete direct +frontend inputs, the C and WW source decoders diagnose each raw NUL and omit it +from their logical character streams before token recovery. Thus a NUL cannot +split an escape, identifier, number or suffix, operator, comment delimiter, or +EOF boundary into a stage-dependent second error. Test loading maintains a +stage-equal source-diagnostic stream before import discovery, graph actions, +producers, test execution, accounting, result output, or retained publication. +An invalid selected test request emits only the existing attributable +`FAIL\n`; it does not construct a variant, generated main, test child, or new +persistent generation. Wrong-target exclusion remains first, so an excluded +file with a raw NUL has no diagnostic or persistence effect. A `\\x00` escape +remains a valid literal value. This rule is limited to raw U+0000; malformed +UTF-8 and the independent per-source BOM rule retain their existing, separate +contracts. + After that eligibility boundary and the coordinator's required package-clause classification and production `@test` validation parses, the delegated loader performs selected-basename Go 1.26.5 simple-fold preflight before its graph diff --git a/internal/wwpackage/package.ww b/internal/wwpackage/package.ww index d24dd37f..01f68563 100644 --- a/internal/wwpackage/package.ww +++ b/internal/wwpackage/package.ww @@ -624,6 +624,16 @@ fn pkgfailpath(path: str, reason: str) void = { pkgputln(os.STDERR_FILENO, reason); }; +fn pkgfailsource(path: str, line: i32, col: i32, reason: str) void = { + pkgput(os.STDERR_FILENO, path); + pkgput(os.STDERR_FILENO, ":"); + pkgput(os.STDERR_FILENO, strconv.i32tos(line, strconv.base.DEC)); + pkgput(os.STDERR_FILENO, ":"); + pkgput(os.STDERR_FILENO, strconv.i32tos(col, strconv.base.DEC)); + pkgput(os.STDERR_FILENO, ": error: "); + pkgputln(os.STDERR_FILENO, reason); +}; + fn pkgusage() void = { pkgput(os.STDERR_FILENO, "usage: wwtest package [-c] [-S] [-list] [-j N] [-I DIR] [-L DIR] [-l LIB] [-w DIR] [-run|-filter GLOB] [-timeout-ms=N] [DIR | DIR/... ...] [-- GLOB ...]\n"); @@ -689,6 +699,27 @@ fn pkgident(c: u8) bool = { return c >= '0' && c <= '9'; }; +fn pkgnulerrors(path: str, src: str) bool = { + let i: i32 = 0; + let ln: i32 = 1; + let cl: i32 = 1; + let found: bool = false; + for (i < src.len) { + if (src[i] == 0u8) { + pkgfailsource(path, ln, cl, "invalid NUL character"); + found = true; + }; + if (src[i] == '\n') { + ln += 1; + cl = 1; + } else { + cl += 1; + }; + i += 1; + }; + return found; +}; + fn pkgskipspace(src: str, start: i32) i32 = { let i: i32 = start; for (i < src.len) { @@ -2260,7 +2291,14 @@ export fn packagecommand(args: []str) int = { }; let body: str; let pn: str; - if (!pkgread(ds.paths[i], &body) || !pkgclause(body, &pn)) { + if (!pkgread(ds.paths[i], &body)) { + pkgfailpath(ds.paths[i], "invalid or missing package clause"); + return pkgteststatusfail(explicitstatus, compileonly); + }; + if (pkgnulerrors(ds.paths[i], body)) { + return pkgteststatusfail(explicitstatus, compileonly); + }; + if (!pkgclause(body, &pn)) { pkgfailpath(ds.paths[i], "invalid or missing package clause"); return pkgteststatusfail(explicitstatus, compileonly); }; diff --git a/lib/ww/syntax/lex.ww b/lib/ww/syntax/lex.ww index 1103e42f..b424d3dc 100644 --- a/lib/ww/syntax/lex.ww +++ b/lib/ww/syntax/lex.ww @@ -51,6 +51,7 @@ export type lex = struct { line: i32, col: i32, errs: i32, + nulcount: u64, // a `//ww:module-reset` directive was seen in the last skipped run; // lexnext emits TK_MODRESET before the next real token (#16 opt-B). modreset: i32, @@ -81,6 +82,7 @@ export fn lexinit(l: *lex, file: str, src: *u8, len: u64) void = { l.line = 1; l.col = 1; l.errs = 0; + l.nulcount = 0u64; l.modreset = 0; l.modpathset = 0; l.modresetpathset = 0; @@ -95,34 +97,101 @@ fn srcb(l: *lex, off: u64) i32 = { return b: i32; }; +fn lskipnul(l: *lex) void = { + for (l.lpos < l.srclen) { + if (srcb(l, l.lpos) != 0) { break; }; + let np: pos; + np.file = l.file; + np.line = l.line; + np.col = l.col; + l.lpos += 1u64; + l.col += 1; + errat(l, &np, "invalid NUL character"); + l.nulcount += 1u64; + }; +}; + +// Return the raw offset of a logical byte lookahead. Raw NUL bytes do not +// occupy a slot in the token stream: Go's source.nextch diagnoses them and +// immediately resumes decoding at the following character. +fn lrawoff(l: *lex, ahead0: u64) u64 = { + let p: u64 = l.lpos; + let ahead: u64 = ahead0; + for (true) { + for (p < l.srclen) { + if (srcb(l, p) != 0) { break; }; + p += 1u64; + }; + if (ahead == 0u64 || p >= l.srclen) { return p; }; + p += 1u64; + ahead -= 1u64; + }; +}; + fn lpeek(l: *lex, ahead: u64) i32 = { - let p: u64 = l.lpos + ahead; - if (p >= l.srclen) { return -1; }; - if (bomat(l.src, l.srclen, p)) { return 0xFEFF; }; - return srcb(l, p); + // Drain NUL at the current decoder position even when it is followed by + // EOF; otherwise a trailing NUL could disappear without a diagnostic. + for (true) { + if (l.lpos >= l.srclen) { return -1; }; + let c: i32 = srcb(l, l.lpos); + if (c == 0) { lskipnul(l); continue; }; + if (ahead == 0u64) { + if (bomat(l.src, l.srclen, l.lpos)) { return 0xFEFF; }; + return c; + }; + let p: u64 = lrawoff(l, ahead); + if (p >= l.srclen) { return -1; }; + if (bomat(l.src, l.srclen, p)) { return 0xFEFF; }; + return srcb(l, p); + }; }; fn lget(l: *lex) i32 = { - if (l.lpos >= l.srclen) { return -1; }; - if (bomat(l.src, l.srclen, l.lpos)) { - let bp: pos; - bp.file = l.file; - bp.line = l.line; - bp.col = l.col; - l.lpos += 3u64; - l.col += 3; - errat(l, &bp, "invalid BOM in the middle of the file"); - return 0xFEFF; + for (true) { + if (l.lpos >= l.srclen) { return -1; }; + let c: i32 = srcb(l, l.lpos); + if (c == 0) { lskipnul(l); continue; }; + if (bomat(l.src, l.srclen, l.lpos)) { + let bp: pos; + bp.file = l.file; + bp.line = l.line; + bp.col = l.col; + l.lpos += 3u64; + l.col += 3; + errat(l, &bp, "invalid BOM in the middle of the file"); + return 0xFEFF; + }; + l.lpos += 1u64; + if (c == '\n') { + l.line += 1; + l.col = 1; + } else { + l.col += 1; + }; + return c; }; - let c: i32 = srcb(l, l.lpos); - l.lpos += 1u64; - if (c == '\n') { - l.line += 1; - l.col = 1; - } else { - l.col += 1; +}; + +// Copy one raw source span into token text while omitting diagnosed NUL +// bytes. This keeps keyword, numeric, suffix, and directive recovery on the +// same logical character stream as lpeek/lget. +fn lexspan(l: *lex, begin: u64, end: u64) str = { + let buf: []u8 = alloc([], end - begin + 1u64)!; + let i: u64 = begin; + let j: u64 = 0u64; + for (i < end) { + let b: i32 = srcb(l, i); + if (b != 0) { + buf[j] = b: u8; + j += 1u64; + }; + i += 1u64; }; - return c; + buf[j] = 0u8; + let out: str; + out.ptr = buf.ptr; + out.len = j: i32; + return out; }; fn curpos(l: *lex, out: *pos) void = { @@ -146,6 +215,81 @@ fn errat(l: *lex, p: *pos, msg: str) void = { l.errs += 1; }; +// Consume one // body, then classify the driver's internal module directive +// from its logical (NUL-filtered) text. A single moving lexer cursor keeps long +// generated module paths linear. The trailing newline remains for skipws. +fn linecomment(l: *lex) void = { + let begin: u64 = l.lpos; + let nulbegin: u64 = l.nulcount; + for (true) { + let c: i32 = lpeek(l, 0u64); + if (c < 0 || c == '\n') { break; }; + lget(l); + }; + let end: u64 = l.lpos; + let body: str; + if (l.nulcount == nulbegin) { + body.ptr = l.src + begin; + body.len = (end - begin): i32; + } else { + body = lexspan(l, begin, end); + }; + + let pre: str = "ww:module"; + if (!strings.hasprefix(body, pre) || body.len == pre.len) { return; }; + let i: i32 = pre.len; + if (body[i] == '-') { + let rest: str = "-reset"; + if (body.len - i < rest.len) { return; }; + let j: i32 = 0; + for (j < rest.len) { + if (body[i + j] != rest[j]) { return; }; + j += 1; + }; + i += rest.len; + if (i == body.len) { + l.modreset = 1; + l.modpathset = 0; // #9: reset supersedes a pending path + return; + }; + if (body[i] != ' ' && body[i] != '\t') { return; }; + for (i < body.len && (body[i] == ' ' || body[i] == '\t')) { + i += 1; + }; + let s: i32 = i; + for (i < body.len && body[i] != '\r' && body[i] != ' ' + && body[i] != '\t') { + i += 1; + }; + l.modreset = 1; + l.modpathset = 0; // #9: see above + if (i > s) { + let view: str; + view.ptr = body.ptr + (s: u64); + view.len = i - s; + l.modresetpath = strings.dup(view); + l.modresetpathset = 1; + }; + return; + }; + if (body[i] != ' ' && body[i] != '\t') { return; }; + for (i < body.len && (body[i] == ' ' || body[i] == '\t')) { + i += 1; + }; + let s: i32 = i; + for (i < body.len && body[i] != '\r' && body[i] != ' ' + && body[i] != '\t') { + i += 1; + }; + if (i > s) { + let view: str; + view.ptr = body.ptr + (s: u64); + view.len = i - s; + l.modpath = strings.dup(view); + l.modpathset = 1; + }; +}; + fn skipws(l: *lex) bool = { for (true) { let c: i32 = lpeek(l, 0u64); @@ -158,111 +302,9 @@ fn skipws(l: *lex) bool = { let c2: i32 = lpeek(l, 1u64); if (c2 == '/') { lget(l); lget(l); - // #16 opt-B: recognize the driver's curmod-reset - // boundary directive `//ww:module-reset` (whole - // line) and flag it; lexnext emits TK_MODRESET. - // The body is then skipped like any comment. - // Mirrors cstage lex.c skipws. Compare via lpeek - // (no consume) so the skip loop below is unchanged. - let pre: str = "ww:module"; - let di: i32 = 0; - let matched: bool = true; - for (di < pre.len) { - if (lpeek(l, di: u64) != pre[di]: i32) { - matched = false; break; - }; - di += 1; - }; - if (matched) { - let nx: i32 = lpeek(l, pre.len: u64); - if (nx == '-') { - let rest: str = "-reset"; - let rj: i32 = 0; - let rm: bool = true; - for (rj < rest.len) { - if (lpeek(l, (pre.len + rj): u64) - != rest[rj]: i32) { - rm = false; break; - }; - rj += 1; - }; - if (rm) { - let af: i32 = lpeek(l, - (pre.len + rest.len): u64); - if (af == '\n') { l.modreset = 1; l.modpathset = 0; } // #9: reset supersedes pending path (empty module body) - else { if (af < 0) { l.modreset = 1; l.modpathset = 0; } // #9: see above - else { if (af == ' ' || af == '\t') { - // `//ww:module-reset ` — sep - // primary body tagged by its full - // dotted import path (#57). - let k: u64 = - (pre.len + rest.len): u64; - for (true) { - let sc: i32 = lpeek(l, k); - if (sc == ' ' || sc == '\t') { - k += 1u64; continue; - }; - break; - }; - let s0: u64 = k; - for (true) { - let pc: i32 = lpeek(l, k); - if (pc < 0) { break; }; - if (pc == '\n' || pc == '\r' - || pc == ' ' - || pc == '\t') { - break; - }; - k += 1u64; - }; - l.modreset = 1; - l.modpathset = 0; // #9: see above — clear pending path - if (k > s0) { - let view: str; - view.ptr = - l.src + l.lpos + s0; - view.len = (k - s0): i32; - l.modresetpath = - strings.dup(view); - l.modresetpathset = 1; - }; - }; }; }; - }; - } else { if (nx == ' ' || nx == '\t') { - // `//ww:module ` — M1 import boundary. - let k: u64 = pre.len: u64; - for (true) { - let sc: i32 = lpeek(l, k); - if (sc == ' ' || sc == '\t') { - k += 1u64; continue; - }; - break; - }; - let s0: u64 = k; - for (true) { - let pc: i32 = lpeek(l, k); - if (pc < 0) { break; }; - if (pc == '\n' || pc == '\r' - || pc == ' ' || pc == '\t') { - break; - }; - k += 1u64; - }; - if (k > s0) { - let view: str; - view.ptr = l.src + l.lpos + s0; - view.len = (k - s0): i32; - l.modpath = strings.dup(view); - l.modpathset = 1; - }; - }; }; - }; - for (true) { - let cx: i32 = lpeek(l, 0u64); - if (cx < 0) { return false; }; - if (cx == '\n') { break; }; - lget(l); - }; + // #16 opt-B: classify the driver's whole-line internal + // module boundary after consuming it once. + linecomment(l); continue; }; if (c2 == '*') { @@ -395,6 +437,38 @@ fn escape(l: *lex, out: *i32) bool = { return false; }; +fn lextypesuffix(l: *lex, out: *str) bool = { + let got: [4]u8; + let n: u64 = 0u64; + for (n < 4u64) { + let c: i32 = lpeek(l, n); + if (c < 0 || !isidpart(c: rune)) { break; }; + got[n] = c: u8; + n += 1u64; + }; + if (n == 0u64 || n > 3u64) { return false; }; + let tail: i32 = lpeek(l, n); + if (tail >= 0) { + if (isidpart(tail: rune)) { return false; }; + }; + let names: []str = ["i8", "i16", "i32", "i64", + "u8", "u16", "u32", "u64", "f32", "f64"]; + let i: i32 = 0; + for (i < names.len) { + if (names[i].len: u64 == n) { + let same: bool = true; + let j: i32 = 0; + for (j < names[i].len) { + if (names[i][j] != got[j]) { same = false; break; }; + j += 1; + }; + if (same) { *out = names[i]; return true; }; + }; + i += 1; + }; + return false; +}; + fn scandecimalrun(l: *lex) void = { for (true) { let c: i32 = lpeek(l, 0u64); @@ -459,6 +533,7 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { out.line = start.line; out.col = start.col; let begin: u64 = l.lpos; + let nulbegin: u64 = l.nulcount; let base: i32 = 10; let isfloat: bool = false; @@ -507,11 +582,16 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { }; }; - let n: u64 = l.lpos - begin; - let view: str; - view.ptr = l.src + begin; - view.len = n: i32; - out.text = strings.dup(view); + let rawend: u64 = l.lpos; + if (l.nulcount == nulbegin) { + let view: str; + view.ptr = l.src + begin; + view.len = (rawend - begin): i32; + out.text = strings.dup(view); + } else { + out.text = lexspan(l, begin, rawend); + }; + let n: u64 = out.text.len: u64; if (isfloat) { out.kind = tkind.TK_FLOAT; @@ -522,7 +602,7 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { let i: u64 = 0u64; let j: u64 = 0u64; for (i < n) { - let b: u8 = l.src[begin + i]; + let b: u8 = out.text[i]; if (b != '_') { clean[j] = b; j += 1u64; @@ -561,7 +641,7 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { let pu: *u64 = (&fv): *u64; out.uval = *pu; } else { - let digs: *u8 = l.src + begin; + let digs: *u8 = out.text.ptr; let dn: u64 = n; if (base != 10) { digs = digs + 2u64; @@ -578,49 +658,11 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { let pc: i32 = lpeek(l, 0u64); if (pc >= 0) { if (isidstart(pc: rune)) { - let sb: u64 = l.lpos; - let sc: i32 = l.col; - for (true) { - let cc: i32 = lpeek(l, 0u64); - if (cc < 0) { break; }; - if (!isidpart(cc: rune)) { break; }; - lget(l); - }; - let sl: u64 = l.lpos - sb; - let p: *u8 = l.src + sb; - let isok: bool = false; - if (sl == 2u64) { - if (p[0] == 'i') { - if (p[1] == '8') { isok = true; }; // i8 - }; - if (p[0] == 'u') { - if (p[1] == '8') { isok = true; }; // u8 - }; - }; - if (sl == 3u64) { - if (p[0] == 'i') { - if (p[1] == '1') { if (p[2] == '6') { isok = true; }; }; // i16 - if (p[1] == '3') { if (p[2] == '2') { isok = true; }; }; // i32 - if (p[1] == '6') { if (p[2] == '4') { isok = true; }; }; // i64 - }; - if (p[0] == 'u') { - if (p[1] == '1') { if (p[2] == '6') { isok = true; }; }; - if (p[1] == '3') { if (p[2] == '2') { isok = true; }; }; - if (p[1] == '6') { if (p[2] == '4') { isok = true; }; }; - }; - if (p[0] == 'f') { - if (p[1] == '3') { if (p[2] == '2') { isok = true; }; }; // f32 - if (p[1] == '6') { if (p[2] == '4') { isok = true; }; }; // f64 - }; - }; - if (isok) { - let view: str; - view.ptr = p; - view.len = sl: i32; - out.tsuffix = strings.dup(view); - } else { - l.lpos = sb; - l.col = sc; + let suffix: str; + if (lextypesuffix(l, &suffix)) { + let i: i32 = 0; + for (i < suffix.len) { lget(l); i += 1; }; + out.tsuffix = strings.dup(suffix); }; }; }; @@ -628,14 +670,24 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { fn lexident(l: *lex, start: *pos, out: *tok) void = { let begin: u64 = l.lpos; + let nulbegin: u64 = l.nulcount; for (true) { let c: i32 = lpeek(l, 0u64); if (c < 0) { break; }; if (!isidpart(c: rune)) { break; }; lget(l); }; - let n: u64 = l.lpos - begin; - let p: *u8 = l.src + begin; + let text: str; + if (l.nulcount == nulbegin) { + let view: str; + view.ptr = l.src + begin; + view.len = (l.lpos - begin): i32; + text = strings.dup(view); + } else { + text = lexspan(l, begin, l.lpos); + }; + let n: u64 = text.len: u64; + let p: *u8 = text.ptr; out.file = start.file; out.line = start.line; out.col = start.col; @@ -643,10 +695,7 @@ fn lexident(l: *lex, start: *pos, out: *tok) void = { if (n == 1u64) { if (p[0] == '_') { out.kind = tkind.TK_UNDER; - let view: str; - view.ptr = p; - view.len = n: i32; - out.text = strings.dup(view); + out.text = text; return; }; }; @@ -656,10 +705,7 @@ fn lexident(l: *lex, start: *pos, out: *tok) void = { } else { out.kind = tkind.TK_IDENT; }; - let view: str; - view.ptr = p; - view.len = n: i32; - out.text = strings.dup(view); + out.text = text; }; fn lexstr(l: *lex, start: *pos, out: *tok) void = { diff --git a/lib/ww/syntax/tok_test.ww b/lib/ww/syntax/tok_test.ww index 0719e851..7be27e06 100644 --- a/lib/ww/syntax/tok_test.ww +++ b/lib/ww/syntax/tok_test.ww @@ -339,6 +339,30 @@ fn checkfloat(src: str, want: u64) void = { assert(!(t.text != "bar")); }; +@test fn long_module_directive_is_linear() void = { + let pre: str = "//ww:module "; + let tail: str = "\nfn"; + let pathlen: i32 = 16384; + let total: i32 = pre.len + pathlen + tail.len; + let src: []u8 = alloc([], total: u64)!; + src.len = total; + let i: i32 = 0; + for (i < pre.len) { src[i] = pre[i]; i += 1; }; + for (i < pre.len + pathlen) { src[i] = 'a': u8; i += 1; }; + let j: i32 = 0; + for (j < tail.len) { src[i] = tail[j]; i += 1; j += 1; }; + let l: syntax.lex; + syntax.lexinit(&l, "t", src.ptr, src.len: u64); + let t: syntax.tok; + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_MODPATH); + assert(t.text.len == pathlen); + assert(t.text[0] == 'a' && t.text[pathlen - 1] == 'a'); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN); + assert(l.errs == 0); +}; + fn bomsource(dst: *u8, before: str, after: str) u64 = { let n: u64 = 0u64; let i: i32 = 0; diff --git a/test/package/package_test.ww b/test/package/package_test.ww index a67979f3..812c6480 100644 --- a/test/package/package_test.ww +++ b/test/package/package_test.ww @@ -130,6 +130,58 @@ fn rewritemidbomfile(path: str, before: str, after: str) void = { putbomfile(path, before, after, true); }; +fn putnulfile(path: str, before: str, after: str, rewrite: bool) void = { + let flags: os.flag = os.flag.WRONLY; + if (rewrite) { flags |= os.flag.TRUNC; } + else { flags |= os.flag.CREATE | os.flag.EXCL; }; + let fd: i32 = os.open(path, flags, 384i32); + assert(fd >= 0); + match (os.writeall(fd, before.ptr, before.len: u64)) { + case let n: i64 => assert(n == before.len: i64); + case let e: os.oserror => abort("write before NUL failed"); + }; + let nul: [1]u8; + nul[0] = 0u8; + match (os.writeall(fd, nul.ptr, 1u64)) { + case let n: i64 => assert(n == 1i64); + case let e: os.oserror => abort("write NUL failed"); + }; + match (os.writeall(fd, after.ptr, after.len: u64)) { + case let n: i64 => assert(n == after.len: i64); + case let e: os.oserror => abort("write after NUL failed"); + }; + assert(os.close(fd) == 0); +}; + +fn writenulfile(path: str, before: str, after: str) void = { + putnulfile(path, before, after, false); +}; + +fn rewritenulfile(path: str, before: str, after: str) void = { + putnulfile(path, before, after, true); +}; + +fn writedoublenulfile(path: str, before: str, after: str) void = { + let fd: i32 = os.open(path, + os.flag.WRONLY | os.flag.CREATE | os.flag.EXCL, 384i32); + assert(fd >= 0); + match (os.writeall(fd, before.ptr, before.len: u64)) { + case let n: i64 => assert(n == before.len: i64); + case let e: os.oserror => abort("write before NUL pair failed"); + }; + let pair: [2]u8; + pair[0] = 0u8; pair[1] = 0u8; + match (os.writeall(fd, pair.ptr, 2u64)) { + case let n: i64 => assert(n == 2i64); + case let e: os.oserror => abort("write NUL pair failed"); + }; + match (os.writeall(fd, after.ptr, after.len: u64)) { + case let n: i64 => assert(n == after.len: i64); + case let e: os.oserror => abort("write after NUL pair failed"); + }; + assert(os.close(fd) == 0); +}; + fn writeexecutable(path: str, content: str) void = { let fd: i32 = os.open(path, os.flag.WRONLY | os.flag.CREATE | os.flag.EXCL, 448i32); @@ -16968,3 +17020,479 @@ fn runtimepath(relative: str) str = { && !directoryhasnew(root)); clean(root); }; + +// Go 1.26.5's physical-source reader rejects U+0000 before token context can +// turn it into comment or literal payload. Selected roots, dependencies, and +// every test-source role share that rule; target-excluded files do not. +@test fn raw_nul_is_rejected_in_every_selected_source() void = { + let root: str = fresh(); + let source: str = strings.concat(root, "/source"); + mkdirall(source); + + // Direct frontend ownership and exact C/WW recovery parity. + let leading: str = strings.concat(root, "/leading.ww"); + let comment: str = strings.concat(root, "/comment.ww"); + let stringlit: str = strings.concat(root, "/string.ww"); + let runelit: str = strings.concat(root, "/rune.ww"); + let between: str = strings.concat(root, "/between.ww"); + let escapechar: str = strings.concat(root, "/escape-char.ww"); + let escapedigit: str = strings.concat(root, "/escape-digit.ww"); + let identifier: str = strings.concat(root, "/identifier.ww"); + let lineend: str = strings.concat(root, "/line-end.ww"); + let blockend: str = strings.concat(root, "/block-end.ww"); + let operator: str = strings.concat(root, "/operator.ww"); + let number: str = strings.concat(root, "/number.ww"); + let suffix: str = strings.concat(root, "/suffix.ww"); + let trailing: str = strings.concat(root, "/trailing.ww"); + writenulfile(leading, "", + "package main;\nfn main() void = {};\n"); + writenulfile(comment, "package main;\n// NUL: ", + " here\nfn main() void = {};\n"); + writenulfile(stringlit, + "package main;\nfn main() void = { let value: str = \"NUL: ", + "\"; };\n"); + writenulfile(runelit, + "package main;\nfn main() void = { let value: rune = '", + "A'; };\n"); + writenulfile(between, "package main;\n", + "\nfn main() void = {};\n"); + writenulfile(escapechar, + "package main;\nfn main() void = { let value: str = \"\\", + "n\"; };\n"); + writenulfile(escapedigit, + "package main;\nfn main() void = { let value: str = \"\\x0", + "0\"; };\n"); + writenulfile(identifier, "pack", + "age main;\nfn main() void = {};\n"); + writenulfile(lineend, "package main;\n// comment", + "\nfn main() void = {};\n"); + writenulfile(blockend, "package main;\n/* end *", + "/\nfn main() void = {};\n"); + writenulfile(operator, + "package main;\nfn main() void = { let value: bool = true =", + "= true; };\n"); + writenulfile(number, + "package main;\nfn main() void = { let value: i32 = 1", + "_0; };\n"); + writenulfile(suffix, + "package main;\nfn main() void = { let value: i32 = 1i", + "32; };\n"); + writenulfile(trailing, + "package main;\nfn main() void = {};\n", ""); + let direct: []str = [leading, comment, stringlit, runelit, between, + escapechar, escapedigit, identifier, lineend, blockend, operator, + number, suffix, trailing]; + let positions: []str = [":1:1: error: invalid NUL character\n", + ":2:9: error: invalid NUL character\n", + ":2:43: error: invalid NUL character\n", + ":2:39: error: invalid NUL character\n", + ":2:1: error: invalid NUL character\n", + ":2:39: error: invalid NUL character\n", + ":2:41: error: invalid NUL character\n", + ":1:5: error: invalid NUL character\n", + ":2:11: error: invalid NUL character\n", + ":2:9: error: invalid NUL character\n", + ":2:44: error: invalid NUL character\n", + ":2:38: error: invalid NUL character\n", + ":2:39: error: invalid NUL character\n", + ":3:1: error: invalid NUL character\n"]; + let compilers: []str = ["w6c", "w6c_ww"]; + let stages: []str = ["ww", "ww_ww"]; + let tags: []str = ["c", "ww"]; + let di: i32 = 0; + for (di < direct.len) { + let reference: str = ""; + let si: i32 = 0; + for (si < compilers.len) { + let asmout: str = strings.concat(root, "/direct-", tags[si], "-", + boundarypkgname(di), ".s"); + let av: []str = [driver(compilers[si]), "-c", "--command-package", + "-o", asmout, direct[di]]; + let out: commandout; + runcommand(root, strings.concat("nul-direct-", tags[si], "-", + boundarypkgname(di)), av, + (30i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 && has(out.stderr, positions[di])); + assert(occurrences(out.stderr, "invalid NUL character") == 1); + assert(!os.exists(asmout)); + if (si == 0) { reference = strings.dup(out.stderr); } + else { assert(same(reference, out.stderr)); }; + si += 1; + }; + di += 1; + }; + + // Decoder recovery is also exact for repeated bytes and the internal + // module-boundary directive used by package-unit composition. + let repeated: str = strings.concat(root, "/repeated.ww"); + let directive: str = strings.concat(root, "/directive.ww"); + writedoublenulfile(repeated, "pack", + "age main;\nfn main() void = {};\n"); + writenulfile(directive, "//ww:module-reset ma", + "in\npackage main;\nfn main() void = {};\n"); + let recoverypaths: []str = [repeated, directive]; + let recoverycounts: []i32 = [2, 1]; + let recoverypositions: []str = [ + ":1:5: error: invalid NUL character\n", + ":1:21: error: invalid NUL character\n"]; + let ri: i32 = 0; + for (ri < recoverypaths.len) { + let reference: str = ""; + let rsi: i32 = 0; + for (rsi < compilers.len) { + let asmout: str = strings.concat(root, "/recovery-", tags[rsi], + "-", boundarypkgname(ri), ".s"); + let av: []str = [driver(compilers[rsi]), "-c", + "--command-package", "-o", asmout, recoverypaths[ri]]; + let out: commandout; + runcommand(root, strings.concat("nul-recovery-", tags[rsi], "-", + boundarypkgname(ri)), av, + (30i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, recoverypositions[ri]) + && occurrences(out.stderr, + "invalid NUL character") == recoverycounts[ri] + && !os.exists(asmout)); + if (ri == 0) { + assert(has(out.stderr, + ":1:6: error: invalid NUL character\n")); + }; + if (rsi == 0) { reference = strings.dup(out.stderr); } + else { assert(same(reference, out.stderr)); }; + rsi += 1; + }; + ri += 1; + }; + + // An escape denotes a literal value; it is not a raw source NUL. + let escaped: str = strings.concat(root, "/escaped.ww"); + writefile(escaped, strings.concat( + "package main;\n", + "fn main() i32 = { let value: str = \"\\x00\"; return value.len - 1; };\n")); + let escapedasm: []str = [strings.concat(root, "/escaped-c.s"), + strings.concat(root, "/escaped-ww.s")]; + let si: i32 = 0; + for (si < compilers.len) { + let av: []str = [driver(compilers[si]), "-c", "--command-package", + "-o", escapedasm[si], escaped]; + let out: commandout; + runcommand(root, strings.concat("nul-escaped-", tags[si]), av, + (30i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0); + si += 1; + }; + assert(same(readfile(escapedasm[0]), readfile(escapedasm[1]))); + + let dep: str = strings.concat(source, "/dep"); + let app: str = strings.concat(source, "/app"); + let bad: str = strings.concat(source, "/bad"); + let good: str = strings.concat(source, "/good"); + mkdirall(dep); mkdirall(app); mkdirall(bad); mkdirall(good); + writenulfile(strings.concat(dep, "/dep.ww"), + "package dep;\n// NUL: ", + " dependency\nexport fn value() i32 = { return 42; };\n"); + writefile(strings.concat(app, "/main.ww"), strings.concat( + "package main;\nimport dep;\n", + "fn main() i32 = { return dep.value() - 42; };\n")); + writefile(strings.concat(bad, "/a.ww"), + "package main;\nfn helper() i32 = { return 1; };\n"); + writenulfile(strings.concat(bad, "/main.ww"), + "package main;\n// NUL: ", strings.concat( + " root\nimport missing;\n", + "fn main() i32 = { return 0; };\n")); + writefile(strings.concat(good, "/main.ww"), + "package main;\nfn main() i32 = { return 0; };\n"); + writenulfile(strings.concat(good, "/ignored_windows.ww"), + "not package NUL: ", " and not WW\n"); + + // Invalid root and dependency sources stop before graph actions and public + // output. The wrong-target source remains outside the semantic input set. + let goodbins: []str = [strings.concat(root, "/good-c"), + strings.concat(root, "/good-ww")]; + let invaliddiags: []str = ["", ""]; + let rootdiags: []str = ["", ""]; + si = 0; + for (si < stages.len) { + let depwork: str = strings.concat(root, "/dep-work-", tags[si]); + let depout: str = strings.concat(root, "/dep-output-", tags[si]); + mkdirall(depwork); + let depav: []str = [driver(stages[si]), "build", "-w", depwork, + "-I", source, "-o", depout, "app"]; + let out: commandout; + runcommand(root, strings.concat("nul-dependency-", tags[si]), depav, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, "/dep/dep.ww:2:9: error: invalid NUL character\n")); + assert(occurrences(out.stderr, "invalid NUL character") == 1 + && !has(out.stderr, "w6c failed")); + assert(!os.exists(depout) && directoryisempty(depwork)); + invaliddiags[si] = strings.dup(out.stderr); + + let badwork: str = strings.concat(root, "/bad-work-", tags[si]); + let badout: str = strings.concat(root, "/bad-output-", tags[si]); + mkdirall(badwork); + let badav: []str = [driver(stages[si]), "build", "-w", badwork, + "-I", source, "-o", badout, "bad"]; + runcommand(root, strings.concat("nul-root-", tags[si]), badav, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, "/bad/main.ww:2:9: error: invalid NUL character\n")); + assert(!has(out.stderr, "cannot find import") + && occurrences(out.stderr, "invalid NUL character") == 1); + assert(!os.exists(badout) && directoryisempty(badwork)); + rootdiags[si] = strings.dup(out.stderr); + + let goodwork: str = strings.concat(root, "/good-work-", tags[si]); + mkdirall(goodwork); + let goodav: []str = [driver(stages[si]), "build", "-w", goodwork, + "-I", source, "-o", goodbins[si], "good"]; + runcommand(root, strings.concat("nul-wrong-target-", tags[si]), goodav, + (120i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0 + && !directoryhasnew(goodwork) + && !directoryhasfragment(goodwork, ".wwtxn.")); + let runav: []str = [goodbins[si]]; + runcommand(root, strings.concat("nul-good-run-", tags[si]), runav, + time.second, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0); + si += 1; + }; + assert(same(invaliddiags[0], invaliddiags[1]) + && same(rootdiags[0], rootdiags[1]) + && same(readfile(goodbins[0]), readfile(goodbins[1]))); + + // Production, same-package, external-package, and test-only sources all + // reject before product construction, execution, accounting, and retention. + let prodcase: str = strings.concat(source, "/prodcase"); + let samecase: str = strings.concat(source, "/samecase"); + let externalcase: str = strings.concat(source, "/externalcase"); + let onlycase: str = strings.concat(source, "/onlycase"); + let repeatcase: str = strings.concat(source, "/repeatcase"); + mkdirall(prodcase); mkdirall(samecase); mkdirall(externalcase); + mkdirall(onlycase); mkdirall(repeatcase); + writefile(strings.concat(prodcase, "/more.ww"), + "package prodcase;\nfn more() i32 = { return 2; };\n"); + writenulfile(strings.concat(prodcase, "/prod.ww"), + "package prodcase;\n// NUL: ", + " production\nexport fn value() i32 = { return 1; };\n"); + writefile(strings.concat(prodcase, "/same_test.ww"), + "package prodcase;\n@test fn must_not_run() void = { abort(); };\n"); + writefile(strings.concat(samecase, "/prod.ww"), + "package samecase;\nfn value() i32 = { return 1; };\n"); + writenulfile(strings.concat(samecase, "/same_test.ww"), + "package samecase;\n// NUL: ", + " same\n@test fn must_not_run() void = { abort(); };\n"); + writefile(strings.concat(externalcase, "/prod.ww"), + "package externalcase;\nexport fn value() i32 = { return 1; };\n"); + writenulfile(strings.concat(externalcase, "/external_test.ww"), + "package externalcase_test;\n// NUL: ", strings.concat( + " external\nimport externalcase;\n", + "@test fn must_not_run() void = { abort(); };\n")); + writenulfile(strings.concat(onlycase, "/only_test.ww"), "", + "package onlycase;\n@test fn must_not_run() void = { abort(); };\n"); + writedoublenulfile(strings.concat(repeatcase, "/prod.ww"), + "package repeatcase;\n// pair ", + " here\nfn value() i32 = { return 1; };\n"); + let testids: []str = ["prodcase", "samecase", "externalcase", "onlycase", + "repeatcase"]; + let testpositions: []str = ["/prodcase/prod.ww:2:9: error: invalid NUL character\n", + "/samecase/same_test.ww:2:9: error: invalid NUL character\n", + "/externalcase/external_test.ww:2:9: error: invalid NUL character\n", + "/onlycase/only_test.ww:1:1: error: invalid NUL character\n", + "/repeatcase/prod.ww:2:9: error: invalid NUL character\n"]; + let testcounts: []i32 = [1, 1, 1, 1, 2]; + let testdiags: []str = ["", "", "", "", ""]; + si = 0; + for (si < stages.len) { + let ti: i32 = 0; + for (ti < testids.len) { + let testwork: str = strings.concat(root, "/test-work-", tags[si], "-", + boundarypkgname(ti)); + let testout: str = strings.concat(root, "/test-output-", tags[si], "-", + boundarypkgname(ti)); + mkdirall(testwork); + let av: []str = [driver(stages[si]), "test", "-w", testwork, + "-I", source, "-o", testout, testids[ti]]; + let out: commandout; + runcommand(root, strings.concat("nul-test-", tags[si], "-", + boundarypkgname(ti)), av, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(same(out.stdout, "FAIL\n") + && has(out.stderr, testpositions[ti])); + assert(occurrences(out.stderr, + "invalid NUL character") == testcounts[ti] + && !has(out.stdout, "must_not_run") + && !has(out.stdout, "discovered") && !has(out.stdout, "ok ")); + if (ti == 4) { + assert(has(out.stderr, + "/repeatcase/prod.ww:2:10: error: invalid NUL character\n")); + }; + assert(!os.exists(testout) && directoryisempty(testwork)); + if (si == 0) { testdiags[ti] = strings.dup(out.stderr); } + else { assert(same(testdiags[ti], out.stderr)); }; + ti += 1; + }; + si += 1; + }; + + // A warm invalid edit commits nothing and preserves prior public and action + // bytes. Exact restoration reuses the committed generation without a tool. + let warm: str = strings.concat(source, "/warm"); + mkdirall(warm); + let warmpath: str = strings.concat(warm, "/main.ww"); + let warmvalid: str = + "package main;\nfn main() i32 = { return 0; };\n"; + writefile(warmpath, warmvalid); + let wrapper: str = strings.concat(root, "/nul-compiler-wrapper.sh"); + writeexecutable(wrapper, strings.concat( + "#!/bin/sh\n", + "printf 'compile\\n' >> \"$WW_NUL_TRACE\"\n", + "exec \"$WW_NUL_REAL\" \"$@\"\n")); + let commandsuffixes: []str = [".unit.ww", ".wwi", ".s", ".o", ".a", + ".init.unit.ww", ".init.s", ".init.o"]; + si = 0; + for (si < stages.len) { + let warmwork: str = strings.concat(root, "/warm-work-", tags[si]); + let warmout: str = strings.concat(root, "/warm-output-", tags[si]); + let trace: str = strings.concat(root, "/warm-trace-", tags[si]); + mkdirall(warmwork); writefile(trace, ""); + let env: []str = os.getenvs(); + append(env, strings.concat("WW_W6C=", wrapper)); + append(env, strings.concat("WW_NUL_TRACE=", trace)); + append(env, strings.concat("WW_NUL_REAL=", driver(compilers[si]))); + let av: []str = [driver(stages[si]), "build", "-w", warmwork, + "-I", source, "-o", warmout, "warm"]; + let out: commandout; + runcommandenv(root, strings.concat("nul-warm-cold-", tags[si]), av, + env, (120i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0 + && readfile(trace).len != 0); + let refs: []str = alloc([], commandsuffixes.len: u64)!; + let fi: i32 = 0; + for (fi < commandsuffixes.len) { + append(refs, strings.dup(readfile(strings.concat(warmwork, + "/warm", commandsuffixes[fi])))); + fi += 1; + }; + let toolpaths: []str = ["/.wwtool.ww", "/.wwtool.w6c", + "/.wwtool.w6a", "/.wwtool.stamp"]; + let toolrefs: []str = alloc([], toolpaths.len: u64)!; + fi = 0; + for (fi < toolpaths.len) { + append(toolrefs, strings.dup(readfile(strings.concat(warmwork, + toolpaths[fi])))); + fi += 1; + }; + let binref: str = strings.dup(readfile(warmout)); + rewritefile(trace, ""); + rewritenulfile(warmpath, "package main;\n// NUL: ", + " invalid\nfn main() i32 = { return 0; };\n"); + runcommandenv(root, strings.concat("nul-warm-invalid-", tags[si]), av, + env, (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, ":2:9: error: invalid NUL character\n")); + assert(readfile(trace).len == 0 && same(binref, readfile(warmout))); + fi = 0; + for (fi < commandsuffixes.len) { + assert(same(refs[fi], readfile(strings.concat(warmwork, + "/warm", commandsuffixes[fi])))); + fi += 1; + }; + fi = 0; + for (fi < toolpaths.len) { + assert(same(toolrefs[fi], readfile(strings.concat(warmwork, + toolpaths[fi])))); + fi += 1; + }; + assert(!directoryhasnew(warmwork) + && !directoryhasfragment(warmwork, ".wwtxn.")); + rewritefile(warmpath, warmvalid); + runcommandenv(root, strings.concat("nul-warm-restored-", tags[si]), av, + env, (120i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0 + && readfile(trace).len == 0 && same(binref, readfile(warmout))); + fi = 0; + for (fi < commandsuffixes.len) { + assert(same(refs[fi], readfile(strings.concat(warmwork, + "/warm", commandsuffixes[fi])))); + fi += 1; + }; + fi = 0; + for (fi < toolpaths.len) { + assert(same(toolrefs[fi], readfile(strings.concat(warmwork, + toolpaths[fi])))); + fi += 1; + }; + si += 1; + }; + + // Source validation is request-local under overlapping valid and invalid + // builds, including independent work, outputs, diagnostics, and cleanup. + let pcwork: str = strings.concat(root, "/parallel-c-work"); + let pwwork: str = strings.concat(root, "/parallel-ww-work"); + let pcout: str = strings.concat(root, "/parallel-c-output"); + let pwout: str = strings.concat(root, "/parallel-ww-output"); + mkdirall(pcwork); mkdirall(pwwork); + let pcav: []str = [driver("ww"), "build", "-w", pcwork, + "-I", source, "-o", pcout, "good"]; + let pwav: []str = [driver("ww_ww"), "build", "-w", pwwork, + "-I", source, "-o", pwout, "bad"]; + let pc: exec.command; + pc.path = pcav[0]; pc.argv = pcav; pc.env = os.getenvs(); pc.dir = repo(); + pc.stdoutpath = strings.concat(root, "/parallel-c.stdout"); + pc.stderrpath = strings.concat(root, "/parallel-c.stderr"); + pc.deadline = time.add(time.now(time.clock.monotonic), + (120i64 * (time.second: i64)): time.duration); + pc.grace = (100i64 * (time.millisecond: i64)): time.duration; + let pw: exec.command; + pw.path = pwav[0]; pw.argv = pwav; pw.env = os.getenvs(); pw.dir = repo(); + pw.stdoutpath = strings.concat(root, "/parallel-ww.stdout"); + pw.stderrpath = strings.concat(root, "/parallel-ww.stderr"); + pw.deadline = time.add(time.now(time.clock.monotonic), + (120i64 * (time.second: i64)): time.duration); + pw.grace = (100i64 * (time.millisecond: i64)): time.duration; + let pcp: exec.process; + let pwp: exec.process; + exec.start(&pcp, &pc); exec.start(&pwp, &pw); + let pcdone: bool = false; + let pwdone: bool = false; + for (!pcdone || !pwdone) { + if (!pcdone) { pcdone = exec.poll(&pcp); }; + if (!pwdone) { pwdone = exec.poll(&pwp); }; + if (!pcdone || !pwdone) { + time.sleep(time.millisecond, time.clock.monotonic); + }; + }; + assert(pcp.result.errno == 0 && pcp.result.cleanuperrno == 0 + && pcp.result.termination == exec.termination.EXIT + && pcp.result.code == 0); + assert(pwp.result.errno == 0 && pwp.result.cleanuperrno == 0 + && pwp.result.termination == exec.termination.EXIT + && pwp.result.code == 1); + assert(readfile(pc.stdoutpath).len == 0 && readfile(pc.stderrpath).len == 0); + assert(readfile(pw.stdoutpath).len == 0 + && has(readfile(pw.stderrpath), "invalid NUL character")); + assert(same(readfile(pcout), readfile(goodbins[0])) && !os.exists(pwout)); + assert(directoryisempty(pwwork) && !directoryhasnew(pcwork) + && !directoryhasfragment(pcwork, ".wwtxn.")); + let runav: []str = [pcout]; + let out: commandout; + runcommand(root, "nul-parallel-run", runav, time.second, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0); + assert(!directoryhasfragment(root, ".wwtxn.") + && !directoryhasfragment(root, ".install") + && !directoryhasnew(root)); + clean(root); +}; diff --git a/test/wcc/100_lex.c b/test/wcc/100_lex.c index 969b6444..3a4990d5 100644 --- a/test/wcc/100_lex.c +++ b/test/wcc/100_lex.c @@ -18,11 +18,11 @@ toklit(Arena *a, Tok t) } static int -runrow(const char *src, const char *expect) +runrown(const char *src, size_t len, const char *expect, int expecterrs) { Arena *a = newarena(); Lex l; - lexinit(&l, a, "", src, strlen(src)); + lexinit(&l, a, "", src, len); char *got = amalloc(a, 1); got[0] = '\0'; @@ -48,15 +48,23 @@ runrow(const char *src, const char *expect) got[n] = '\0'; } - int ok = strcmp(got, expect) == 0; + int ok = strcmp(got, expect) == 0 + && (expecterrs < 0 || l.errs == expecterrs); if (!ok) { - fprintf(stderr, "lex mismatch:\n src: %s\n" - " want: %s\n got: %s\n", src, expect, got); + fprintf(stderr, "lex mismatch:\n want: %s (%d errors)\n" + " got: %s (%d errors)\n", expect, expecterrs, got, + l.errs); } freearena(a); return ok; } +static int +runrow(const char *src, const char *expect) +{ + return runrown(src, strlen(src), expect, -1); +} + struct row { const char *src, *expect; }; static const struct row rows[] = { @@ -133,11 +141,11 @@ static const struct row rows[] = { * not the token stream. The verbatim Hare messages live at lexunicode * in cmd/wcc/lex.c. */ static int -runerr(const char *src) +runerrn(const char *src, size_t len) { Arena *a = newarena(); Lex l; - lexinit(&l, a, "", src, strlen(src)); + lexinit(&l, a, "", src, len); for (;;) { Tok t = lexnext(&l); if (t.kind == TK_EOF) @@ -151,6 +159,12 @@ runerr(const char *src) return ok; } +static int +runerr(const char *src) +{ + return runerrn(src, strlen(src)); +} + static int runrewind(void) { @@ -205,6 +219,93 @@ static const char *const errrows[] = { "'\\uDFFF'", /* top of the UTF-16 surrogate range */ }; +static const struct { + const char *src; + size_t len; + const char *expect; + int errs; +} nulrows[] = { + { "\0package main;", sizeof "\0package main;" - 1, + "package IDENT(main) ;", 1 }, + { "pack\0age main;", sizeof "pack\0age main;" - 1, + "package IDENT(main) ;", 1 }, + { "1\0_0", sizeof "1\0_0" - 1, "INT(10)", 1 }, + { "1f\0oo", sizeof "1f\0oo" - 1, "INT(1) IDENT(foo)", 1 }, + { "=\0=", sizeof "=\0=" - 1, "==", 1 }, + { "/\0/ comment\nfn", sizeof "/\0/ comment\nfn" - 1, "fn", 1 }, + { "/* end *\0/ fn", sizeof "/* end *\0/ fn" - 1, "fn", 1 }, + { "\"a\0b\"", sizeof "\"a\0b\"" - 1, "STR(ab)", 1 }, + { "'\0A'", sizeof "'\0A'" - 1, "RUNE(65)", 1 }, + { "fn\0", sizeof "fn\0" - 1, "fn", 1 }, + { "f\0\0n", sizeof "f\0\0n" - 1, "fn", 2 }, +}; + +static int +runescapenul(void) +{ + static const char line[] = "\"\\\0n\""; + static const char hex[] = "\"\\x0" "\0" "0\""; + Arena *a = newarena(); + Lex l; + lexinit(&l, a, "", line, sizeof line - 1); + Tok t = lexnext(&l); + int ok = t.kind == TK_STR && t.tlen == 1 && t.text[0] == '\n' + && l.errs == 1 && lexnext(&l).kind == TK_EOF; + lexinit(&l, a, "", hex, sizeof hex - 1); + t = lexnext(&l); + ok = ok && t.kind == TK_STR && t.tlen == 1 && t.text[0] == '\0' + && l.errs == 1 && lexnext(&l).kind == TK_EOF; + if (!ok) + fprintf(stderr, "NUL escape recovery failed\n"); + freearena(a); + return ok; +} + +static int +runsuffixnul(void) +{ + static const char src[] = "1i" "\0" "8"; + Arena *a = newarena(); + Lex l; + lexinit(&l, a, "", src, sizeof src - 1); + Tok t = lexnext(&l); + int ok = t.kind == TK_INT && t.v.uval == 1 && t.tsuffix != NULL + && strcmp(t.tsuffix, "i8") == 0 && l.errs == 1 + && lexnext(&l).kind == TK_EOF; + if (!ok) + fprintf(stderr, "NUL typed-suffix recovery failed\n"); + freearena(a); + return ok; +} + +static int +runlongdirective(void) +{ + static const char pre[] = "//ww:module "; + static const char tail[] = "\nfn"; + const size_t pathlen = 16384; + const size_t len = sizeof pre - 1 + pathlen + sizeof tail - 1; + char *src = malloc(len); + if (!src) + return 0; + memcpy(src, pre, sizeof pre - 1); + memset(src + sizeof pre - 1, 'a', pathlen); + memcpy(src + sizeof pre - 1 + pathlen, tail, sizeof tail - 1); + Arena *a = newarena(); + Lex l; + lexinit(&l, a, "", src, len); + Tok m = lexnext(&l); + Tok f = lexnext(&l); + int ok = m.kind == TK_MODPATH && m.tlen == pathlen + && m.text[0] == 'a' && m.text[pathlen - 1] == 'a' + && f.kind == TK_FN && l.errs == 0; + if (!ok) + fprintf(stderr, "long module directive failed\n"); + freearena(a); + free(src); + return ok; +} + int main(void) { @@ -225,6 +326,19 @@ main(void) fail++; } } + for (size_t i = 0; i < sizeof nulrows / sizeof nulrows[0]; i++) { + if (!runrown(nulrows[i].src, nulrows[i].len, + nulrows[i].expect, nulrows[i].errs)) { + fprintf(stderr, "NUL row %zu failed\n", i); + fail++; + } + } + if (!runescapenul()) + fail++; + if (!runsuffixnul()) + fail++; + if (!runlongdirective()) + fail++; if (fail) { fprintf(stderr, "%d/%zu lex tests failed\n", fail, sizeof rows / sizeof rows[0]);