From 4069eda942344ff5749b226b954e5e3046a0cf42 Mon Sep 17 00:00:00 2001 From: Hojun-Cho Date: Sat, 22 Aug 2026 03:14:53 +0900 Subject: [PATCH] ww source: reject malformed UTF-8 --- cmd/wcc/lex.c | 118 +++++++++- cmd/wcc/ww.h | 1 + docs/build-system.md | 205 +++++++++++++++++ docs/spec.md | 18 +- docs/test-system-v2.md | 42 ++++ internal/wwpackage/package.ww | 23 +- lib/ww/syntax/lex.ww | 103 ++++++++- lib/ww/syntax/tok_test.ww | 135 +++++++++++ test/package/package_test.ww | 411 ++++++++++++++++++++++++++++++++++ test/wcc/100_lex.c | 120 ++++++++++ 10 files changed, 1146 insertions(+), 30 deletions(-) diff --git a/cmd/wcc/lex.c b/cmd/wcc/lex.c index 31c5acb5..ddb6fa13 100644 --- a/cmd/wcc/lex.c +++ b/cmd/wcc/lex.c @@ -16,6 +16,61 @@ bomat(const char *src, u64 len, u64 pos) && (unsigned char)src[pos + 2] == 0xbf; } +/* Return the width of the valid UTF-8 sequence beginning at pos, or zero. + * This is the same scalar-value partition used by unicode/utf8.DecodeRune: + * overlong encodings, surrogates, values above U+10FFFF, stray continuation + * bytes, and truncated sequences are invalid. */ +static int +utf8seqwidth(const char *src, u64 len, u64 pos) +{ + if (pos >= len) + return 0; + const unsigned char *s = (const unsigned char *)src; + unsigned char c = s[pos]; + if (c < 0x80) + return 1; + if (c >= 0xc2 && c <= 0xdf && len - pos >= 2 + && s[pos + 1] >= 0x80 && s[pos + 1] <= 0xbf) + return 2; + if (len - pos >= 3 && s[pos + 2] >= 0x80 + && s[pos + 2] <= 0xbf) { + unsigned char c1 = s[pos + 1]; + if ((c == 0xe0 && c1 >= 0xa0 && c1 <= 0xbf) + || (c >= 0xe1 && c <= 0xec && c1 >= 0x80 && c1 <= 0xbf) + || (c == 0xed && c1 >= 0x80 && c1 <= 0x9f) + || (c >= 0xee && c <= 0xef && c1 >= 0x80 && c1 <= 0xbf)) + return 3; + } + if (len - pos >= 4 && s[pos + 2] >= 0x80 + && s[pos + 2] <= 0xbf && s[pos + 3] >= 0x80 + && s[pos + 3] <= 0xbf) { + unsigned char c1 = s[pos + 1]; + if ((c == 0xf0 && c1 >= 0x90 && c1 <= 0xbf) + || (c >= 0xf1 && c <= 0xf3 && c1 >= 0x80 && c1 <= 0xbf) + || (c == 0xf4 && c1 >= 0x80 && c1 <= 0x8f)) + return 4; + } + return 0; +} + +/* The lexer deliberately keeps valid non-ASCII source byte-oriented. Decide + * whether one raw byte belongs to a complete valid UTF-8 sequence without + * changing that established token model. A continuation byte is valid only + * when a valid sequence beginning at most three bytes earlier contains it. */ +static int +utf8bytevalid(const char *src, u64 len, u64 pos) +{ + if (utf8seqwidth(src, len, pos) != 0) + return 1; + unsigned char c = (unsigned char)src[pos]; + if (c < 0x80 || c > 0xbf) + return 0; + for (u64 back = 1; back <= 3 && back <= pos; back++) + if (utf8seqwidth(src, len, pos - back) > (int)back) + return 1; + return 0; +} + void lexinit(Lex *l, Arena *a, const char *file, const char *src, u64 len) { @@ -47,16 +102,39 @@ lskipnul(Lex *l) } } -/* Return the raw offset of a logical byte lookahead. Raw NUL bytes do not - * occupy a slot in the token stream: Go's source.nextch diagnoses them and - * immediately resumes decoding at the following character. */ +/* Go 1.26.5 syntax.source.nextch reports and discards one byte whenever + * utf8.DecodeRune returns RuneError with width one. Drain the same malformed + * bytes before they can affect token recovery. */ +static void +lskiputf8(Lex *l) +{ + while (l->pos < l->srclen + && (unsigned char)l->src[l->pos] >= 0x80 + && !utf8bytevalid(l->src, l->srclen, l->pos)) { + Pos p = { l->file, l->line, l->col }; + l->pos++; + l->col++; + errorf(p, "invalid UTF-8 encoding"); + l->errs++; + l->utf8count++; + } +} + +/* Return the raw offset of a logical byte lookahead. Raw NUL and malformed + * UTF-8 bytes do not occupy a slot in the token stream: Go's source.nextch + * diagnoses them and immediately resumes at the following byte. */ static u64 lrawoff(Lex *l, u64 ahead) { u64 p = l->pos; for (;;) { - while (p < l->srclen && l->src[p] == '\0') + while (p < l->srclen) { + unsigned char c = (unsigned char)l->src[p]; + if (c != 0 && (c < 0x80 + || utf8bytevalid(l->src, l->srclen, p))) + break; p++; + } if (ahead == 0 || p >= l->srclen) return p; p++; @@ -77,6 +155,11 @@ lpeek(Lex *l, u64 ahead) lskipnul(l); continue; } + if (c >= 0x80 + && !utf8bytevalid(l->src, l->srclen, l->pos)) { + lskiputf8(l); + continue; + } if (ahead == 0) { if (bomat(l->src, l->srclen, l->pos)) return 0xfeff; @@ -102,6 +185,11 @@ lget(Lex *l) lskipnul(l); continue; } + if (c >= 0x80 + && !utf8bytevalid(l->src, l->srclen, l->pos)) { + lskiputf8(l); + continue; + } if (bomat(l->src, l->srclen, l->pos)) { Pos p = { l->file, l->line, l->col }; l->pos += 3; @@ -128,17 +216,20 @@ lpos(Lex *l) return p; } -/* Copy one raw source span into token text while omitting diagnosed NUL - * bytes. This keeps keyword, numeric, suffix, and directive recovery on the - * same logical character stream as lpeek/lget. */ +/* Copy one raw source span into token text while omitting diagnosed NUL and + * malformed UTF-8 bytes. This keeps keyword, numeric, suffix, and directive + * recovery on the same logical character stream as lpeek/lget. */ static char * lexspan(Lex *l, u64 begin, u64 end, u64 *len) { char *s = amalloc(l->a, end - begin + 1); u64 j = 0; - for (u64 i = begin; i < end; i++) - if (l->src[i] != '\0') + for (u64 i = begin; i < end; i++) { + unsigned char c = (unsigned char)l->src[i]; + if (c != 0 && (c < 0x80 + || utf8bytevalid(l->src, l->srclen, i))) s[j++] = l->src[i]; + } s[j] = '\0'; *len = j; return s; @@ -173,13 +264,14 @@ linecomment(Lex *l) { u64 begin = l->pos; u64 nulbegin = l->nulcount; + u64 utf8begin = l->utf8count; int c; while ((c = lpeek(l, 0)) >= 0 && c != '\n') lget(l); u64 end = l->pos; u64 n; const char *body; - if (l->nulcount == nulbegin) { + if (l->nulcount == nulbegin && l->utf8count == utf8begin) { n = end - begin; body = l->src + begin; } else { @@ -414,6 +506,7 @@ lexnum(Lex *l, Pos start) Tok t = (Tok){ TK_INT, start, NULL, 0, {0}, TK_NONE }; u64 begin = l->pos; u64 nulbegin = l->nulcount; + u64 utf8begin = l->utf8count; int base = 10; int isfloat = 0; int c = lpeek(l, 0); @@ -454,7 +547,7 @@ lexnum(Lex *l, Pos start) u64 rawend = l->pos; u64 n; - if (l->nulcount == nulbegin) { + if (l->nulcount == nulbegin && l->utf8count == utf8begin) { n = rawend - begin; t.text = astrndup(l->a, l->src + begin, n); } else { @@ -511,11 +604,12 @@ lexident(Lex *l, Pos start) { u64 begin = l->pos; u64 nulbegin = l->nulcount; + u64 utf8begin = l->utf8count; while (isidcont(lpeek(l, 0))) lget(l); u64 n; const char *p; - int filtered = l->nulcount != nulbegin; + int filtered = l->nulcount != nulbegin || l->utf8count != utf8begin; if (!filtered) { n = l->pos - begin; p = l->src + begin; diff --git a/cmd/wcc/ww.h b/cmd/wcc/ww.h index af63ea68..46d32029 100644 --- a/cmd/wcc/ww.h +++ b/cmd/wcc/ww.h @@ -203,6 +203,7 @@ struct Lex { Arena *a; /* token-text arena */ int errs; u64 nulcount; /* raw NUL bytes diagnosed by the source decoder */ + u64 utf8count; /* malformed UTF-8 bytes diagnosed/filtered */ int modreset; /* a `//ww:module-reset` directive was seen in * the last skipped run; lexnext emits TK_MODRESET * before the next real token. */ diff --git a/docs/build-system.md b/docs/build-system.md index 3caee2a8..c80204de 100644 --- a/docs/build-system.md +++ b/docs/build-system.md @@ -8828,6 +8828,211 @@ This is command parsing and output presentation only. No persisted-byte contract changes: build workdir format remains `18`, test workdir format remains `19`, and semantic storage format remains `3`. +### 11.45 Implemented selected-source malformed UTF-8 rejection + +Every malformed UTF-8 byte in an eligible selected physical `.ww` source is +rejected at its 1-based physical line and raw-byte column with exactly +`invalid UTF-8 encoding`. The source decoder consumes that byte, omits it from +the logical character stream, and resumes. Consequently, a malformed +multi-byte spelling is diagnosed once for every byte that decodes as U+FFFD +with width one, while a correctly encoded U+FFFD remains valid. Validation is +source-wide: comment and literal contexts do not hide malformed bytes, and an +invalid byte cannot split an identifier, number or suffix, operator, escape, +comment delimiter, package keyword, or import spelling into a different token. + +This section adds only malformed-UTF-8 validation. It does not reopen the +per-source leading-BOM contract in §11.42, raw-U+0000 rejection in §11.43, or +exact output-option parsing in §11.44. BOM, raw NUL, and malformed UTF-8 remain +independent positioned source conditions and are handled in raw-byte order. + +#### Pinned evidence, applicability, and fact classification + +The sole authority is official Go 1.26.5 at +`c19862e5f8415b4f24b189d065ed739517c548ba`: + +- compiler reader `(*source).init`, `(*source).pos`/`error`, and + `(*source).nextch` establish the 1-based byte-positioned decoding boundary in + [`cmd/compile/internal/syntax/source.go`, lines 60–88 and 113–165](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/cmd/compile/internal/syntax/source.go#L60-L165); +- `(*source).nextch` calls `utf8.DecodeRune` at lines 149–150, and a U+FFFD + result of width one reports exactly `invalid UTF-8 encoding`, consumes that + one byte, and resumes at lines 152–154; +- compiler-scanner `TestScanErrors` asserts the positioned malformed byte and + truncated-`EF` regression in + [`cmd/compile/internal/syntax/scanner_test.go`, lines 587–600 and 658](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/cmd/compile/internal/syntax/scanner_test.go#L587-L658); +- compiler testdata requires UTF-8 errors in interpreted and raw strings, + comments, identifiers, and ordinary source in + [`test/nul1.go`, lines 7–52](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/test/nul1.go#L7-L52); +- the independent public scanner corroborates width-one malformed decoding in + [`go/scanner/scanner.go`, lines 63–108](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/go/scanner/scanner.go#L63-L108) + and its literal test at + [`go/scanner/scanner_test.go`, lines 810–811](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/go/scanner/scanner_test.go#L810-L811); +- directory enumeration, filename eligibility, source reading, and test-role + classification are ordered by `Context.Import`, `Context.matchFile`, and + `Context.goodOSArchFile` in + [`go/build/build.go`, lines 859–953 and 1005–1036](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/go/build/build.go#L859-L1036), + [`Context.matchFile`, lines 1438–1509](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/go/build/build.go#L1438-L1509), + and + [`Context.goodOSArchFile`, lines 1980–2027](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/go/build/build.go#L1980-L2027), + while internal/external test variants consume those selected lists in + [`cmd/go/internal/load/test.go`, `TestPackagesAndErrors`, lines 85–102 and 175–240](https://github.com/golang/go/blob/c19862e5f8415b4f24b189d065ed739517c548ba/src/cmd/go/internal/load/test.go#L85-L240). + +Those decoder branches, source-selection branches, and official assertions are +**behavior directly implemented or asserted by pinned Go**. Applying the same +per-selected-physical-source rule to `.ww` inputs, after WW's fixed-target +filename/test-role eligibility and before its package/import interpretation, +is **behavior derived from the pinned implementation**. It honestly applies to +WW's declared UTF-8 source without adding modules, manifests, registries, lock +files, caches, databases, CAS, network resolution, quoted/grouped/dot/general +imports, or a source-level build language. + +Before this change, the following observations were **directly measured WW +behavior**: + +- direct `w6c` and `w6c_ww` accepted comments containing stray continuation + `80`, lead `FF`, overlong `C0 80`, and surrogate `ED A0 80` bytes, exited + zero with empty streams, and emitted byte-identical 89-byte assembly with + SHA-256 + `7385e3ce0107324edc94ee4377c5c2b0262c5b690939cb54f599ed09f9730db8`; +- both direct compilers accepted raw `FF` in a string and emitted + byte-identical 236-byte assembly with SHA-256 + `4dbe79617badb56d541526ca276a9867d0d717ad9aa0c540f629882f87e3f3ad`, + while valid Korean and accented controls remained accepted and + stage-identical; +- both public build stages accepted malformed comments and literals, published + runnable binaries, and built and ran an imported dependency containing + malformed UTF-8; that dependency executable was 4,317 stage-identical bytes + with SHA-256 + `f792da043743cc512c9d9a41deb4cb4af1e42d35e2716fcd5b57355fd7f566eb`; +- malformed production, same-package, external-package, and test-only selected + sources ran successfully in both stages; one same-package compile-only binary + was 112,829 stage-identical bytes with SHA-256 + `108150d071a3ab4264d021d2c3fdc7a94ff71a00153fdf14bfe8e26520cc5da2`; +- a corrupted `package` produced a package-clause diagnostic plus + `unexpected character 0xff` in Cstage but a generic `unexpected character` + in WWstage, with 176-byte versus 171-byte stderr; corrupted imports reached + analogous fallback recovery rather than the pinned decoder diagnostic; +- malformed source before `import nowhere;` allowed missing-import resolution + to win, while a malformed same-package test keyword reached the + coordinator's unpositioned `invalid or missing package clause`; and +- malformed wrong-target and ordinary-build-excluded test files were ignored, + and reverse-created selected files were still diagnosed in byte-sorted name + order. + +No installed host Go result supplies any authority or measurement above. + +#### True ownership and complete four-axis result + +The semantic owners are the source-decoder twins in `cmd/wcc/lex.c` and +`lib/ww/syntax/lex.ww`, plus the selected-physical-source preflight in +`internal/wwpackage/package.ww`. The decoders enforce bytewise recovery for +complete direct and composed compiler inputs. The shared coordinator enforces +the same validation before its textual package-clause classifier and import +discovery, so a coordinator fallback cannot outrank the physical-source error. +The Cstage/WWstage package-unit composers only transport already admitted +source bytes and are not additional owners. + +- **Go-like build:** each eligible selected root, library, or dependency source + is validated before its imports complete the graph or any compiler, + assembler, archiver, linker, install, publication, or runtime action starts. + A source error in a selected root precedes missing, self, cycle, `internal`, + vendor, and imported-command resolution. Wrong-target and test-only files + excluded from ordinary build are not semantic inputs and are not decoded. +- **Go-like test:** production, same-package, external-package, and test-only + selected physical sources are validated before grouping or variant/product + construction. Rejection builds no support action or generated main, starts + no test process, emits no accounting or package `ok` line, and retains or + publishes no executable. An attributable explicit request keeps its existing + command-owned final `FAIL\n` presentation. +- **Go-like package:** each selected physical file owns its positioned errors; + selected filenames retain byte-sorted order. Correctly encoded non-ASCII + content remains legal in WW's permitted comment/literal contexts. Declared + package names, source roles, package conflicts, command/test classification, + and variant boundaries do not change. +- **Go-like import:** a malformed byte is filtered before it can manufacture, + split, or change an import occurrence, qualifier, or edge. Valid spelling, + aliases, file-scoped binding, local/vendor/internal resolution, visibility, + cycle detection, graph order, and initialization remain unchanged. + +Canonical dotted package and import identity remains exact and +case-sensitive. Physical directory, declared name, alias, path leaf, filename, +source bytes, artifact name, output path, and linker order remain loader, +runtime, or presentation metadata only where already specified; none becomes +package, graph, action, symbol, `.wwi`, publication, or persistence identity. + +#### Loading, graph, action, runtime, and diagnostics + +Filename and test-role eligibility occurs first. The shared preflight then +scans eligible selected files in existing byte-sorted order, reporting every +malformed byte in the first invalid physical file in position order before +package-clause classification. Lines and columns advance by raw source bytes. +A legal leading BOM still advances three columns, raw NUL keeps its own exact +diagnostic, and the three source conditions interleave without one being +reclassified as another. + +An invalid source completes no package node, import edge, test variant, +support action, or generated-main action. No compiler, assembler, archiver, +linker, installer, or runtime process is scheduled for that invalid request. +Independent valid siblings and requests keep the established command-global +planning and scheduling rules; validation adds no global state and cannot +cancel or mutate them. Valid loading, graph identity, action order, +initialization, runtime behavior, result order, and output selection are +explicit non-effects. + +Diagnostics use exact text `invalid UTF-8 encoding` with path, 1-based line, +and 1-based raw-byte column. Each width-one malformed decode is consumed and +removed before token recovery, preventing a second stage-specific package, +import, identifier, literal, operator, escape, or EOF interpretation. Complete +direct frontend inputs report all malformed bytes. Public package/test loading +uses the same sequential physical-source preflight and therefore preserves +Cstage/WWstage diagnostic-byte parity and source-before-resolution precedence. +Existing valid-input, BOM, NUL, package, import, and output-option diagnostics +retain their owners and wording. + +#### Publication, persistence, artifacts, and failure lifecycle + +Cold malformed-source rejection creates no synthetic unit, `.wwi`, assembly, +object, archive, executable, retained test binary, result status, capture, or +published output. It leaves no `.new`, `.install`, `.wwtxn.*`, adjacent +`.sepwork`, tool-stage transaction, or scratch residue. Invalid input has no +artifact-byte comparison beyond identical absence. + +Warm rejection commits no generation and preserves every prior unit, +interface, assembly, object, archive, tool record, stamp, executable, retained +binary, and public output byte for byte. Because staging has not begun, the +source branch requires no new rollback mechanism. Restoring the exact valid +source follows ordinary content comparison and may reuse the prior committed +generation. Valid Cstage and WWstage unit, interface, assembly, object, +archive, generated-main, executable, and retained-test bytes keep their +existing byte-identity contract. + +Producer failure, runtime failure, publication-only failure, and cleanup-only +failure remain downstream owners and are not redefined; malformed-source +rejection makes those phases unreachable for the invalid request. Validation +state is source/request-local. Concurrent valid and invalid requests retain +independent workdirs, outputs, captures, diagnostics, processes, and +transactions. The change adds no process, wait, signal, timeout, cancellation, +or interruption boundary, so existing process-group ownership, interruption, +rollback, and cleanup remain unchanged. The recipe-owned fixed +`out/bootstrap` tree is not transaction residue. + +#### Proof, twin parity, and formats + +Focused C and WW lexer proofs cover valid encodings and encoded U+FFFD; +invalid leads and continuations; overlong, surrogate, out-of-range, truncated, +and repeated malformed spellings; token boundaries; and BOM/NUL interaction. +The WW-native `malformed_utf8_is_rejected_in_every_selected_source` observer +owns direct compiler, root/dependency build, source/import precedence, every +test source role, wrong-target selection, cold/warm rejection, exact +restoration and reuse, valid sibling concurrency, artifact absence, residue +cleanup, stage diagnostic parity, and valid-artifact byte parity. Concrete +post-change byte counts, hashes, and gate results are recorded only after +focused and full validation; they are not inferred from the implementation. + +No format bump. This changes invalid-source acceptance and diagnostics only; +the valid persisted-byte contract is unchanged. Build workdir format remains +`18`, test workdir format remains `19`, and semantic storage format remains +`3`. No test-result cache is introduced. + ## 12. Candidate architectures and hard-gate decision Five candidates were developed as coherent systems, not as feature bins. diff --git a/docs/spec.md b/docs/spec.md index bd187f56..715925a9 100644 --- a/docs/spec.md +++ b/docs/spec.md @@ -60,6 +60,19 @@ every other source position, including inside string and rune literals and comments. Apart from that marker rule, the lexer operates on bytes and non-ASCII bytes are legal only inside string and rune literals and comments. +Every malformed UTF-8 byte in an eligible selected physical source produces +one positioned `invalid UTF-8 encoding` error at its 1-based physical line and +raw-byte column. The byte is consumed and omitted from the lexer's logical +character stream before token recovery. A malformed multi-byte spelling is +therefore diagnosed once for each byte that decodes as U+FFFD with width one; +a correctly encoded U+FFFD is valid. Malformed bytes cannot split an +identifier, number or suffix, operator, escape, comment delimiter, package +keyword, or import spelling into different tokens. Filename and test-role +eligibility precede validation, so an excluded physical file contributes no +UTF-8 diagnostic. This rule is independent of the leading-BOM and raw-NUL +rules below and does not make source bytes package, import, graph, action, +artifact, publication, or persistence identity. + Each raw byte `00` (U+0000) is invalid at every physical source position, including in comments and string or rune literal text. It produces one positioned `invalid NUL character` error at that byte's source position and @@ -296,7 +309,10 @@ ImportPath = ident { "." ident } . Production excludes selected `*_test.ww`; test variants classify only those selected test files. An excluded file contributes no declarations, imports, filename collision, package edge, action, export, artifact, initialization, - test, or persistent invalidation. After eligibility, two distinct selected + test, diagnostic, or persistent invalidation. After eligibility, each + selected physical source is validated for malformed UTF-8 and raw NUL before + package-clause or import interpretation, in the existing byte-sorted file + order. After that source preflight, two distinct selected basenames in one canonical directory that are equal under Go 1.26.5 Unicode simple folding are rejected after the coordinator's required package-clause classification and production `@test` validation parses, but before the diff --git a/docs/test-system-v2.md b/docs/test-system-v2.md index 549fdcff..b99ba990 100644 --- a/docs/test-system-v2.md +++ b/docs/test-system-v2.md @@ -459,6 +459,48 @@ remains a valid literal value. This rule is limited to raw U+0000; malformed UTF-8 and the independent per-source BOM rule retain their existing, separate contracts. +After filename and test-role eligibility and before package-clause +classification, every malformed UTF-8 byte in a selected production, +same-package test, external-package test, or test-only source is one positioned +`invalid UTF-8 encoding` error. The shared coordinator scans selected physical +files in existing byte-sorted order and reports all malformed bytes in the +first invalid file before package grouping, import discovery, or delegated +tools. The C and WW source decoders consume each malformed byte, omit it from +their logical streams, and resume, so the byte cannot manufacture or alter a +package token, import spelling, binding, or edge or produce stage-specific +fallback recovery. Correctly encoded U+FFFD and other valid non-ASCII text in +WW's permitted comment and literal contexts remain valid. A malformed +multi-byte spelling is diagnosed byte by byte according to width-one UTF-8 +decodes. Raw NUL and per-source BOM diagnostics remain independent and +interleave with malformed-byte diagnostics in physical byte order. + +Malformed selected source rejects before package/test variant construction, +generated main, compiler, assembler, archiver, linker, runtime, accounting, +retention, or publication. The existing attributable explicit test request +still owns its final `FAIL\n`; no package `ok` line or test-result cache is +created. Cold rejection leaves no unit, `.wwi`, assembly, object, archive, +binary, capture, status, stage, or transaction residue. Warm rejection commits +nothing and preserves the prior generation and public output byte for byte; +restoring the valid bytes follows ordinary exact-content reuse. Producer, +runtime, publication-only, and cleanup failures are unchanged because source +rejection precedes those phases. Validation is source/request-local, adds no +process or signal boundary, cannot contaminate an overlapping valid request, +and leaves interruption and owned-process cleanup with their established +owners. Wrong-target and ordinary-build-excluded test sources remain unread by +this semantic preflight. + +`malformed_utf8_is_rejected_in_every_selected_source` in +`test/package/package_test.ww`, together with focused C and WW lexer coverage, +is the focused proof owner for valid encodings and encoded U+FFFD; stray leads +and continuations; overlong, surrogate, out-of-range, truncated, repeated, and +token-boundary cases; BOM/NUL interaction; root and dependency builds; source +versus import-resolution precedence; all test roles; wrong-target exclusion; +cold and warm rejection; exact restoration/reuse; parallel isolation; +diagnostic and valid-artifact stage parity; and absence of residue. Concrete +post-change measurements and hashes are recorded only after focused and full +validation. The completed BOM, raw-NUL, and exact output-option slices are not +reopened, and canonical dotted package/import identity remains unchanged. + After that eligibility boundary and the coordinator's required package-clause classification and production `@test` validation parses, the delegated loader performs selected-basename Go 1.26.5 simple-fold preflight before its graph diff --git a/internal/wwpackage/package.ww b/internal/wwpackage/package.ww index 80c5362b..eb368f94 100644 --- a/internal/wwpackage/package.ww +++ b/internal/wwpackage/package.ww @@ -699,23 +699,34 @@ fn pkgident(c: u8) bool = { return c >= '0' && c <= '9'; }; -fn pkgnulerrors(path: str, src: str) bool = { +// Validate the selected physical source before package-clause classification. +// Go's source reader consumes malformed UTF-8 one byte at a time; raw NUL is +// the independent ASCII source error. Reporting both in byte order preserves +// the decoder's diagnostic precedence and raw-byte columns. +fn pkgsourceerrors(path: str, src: str) bool = { let i: i32 = 0; let ln: i32 = 1; let cl: i32 = 1; let found: bool = false; for (i < src.len) { - if (src[i] == 0u8) { + let c: u8 = src[i]; + let width: i32 = 1; + if (c >= 128u8) { width = pkgutf8width(src, i); }; + if (c == 0u8) { pkgfailsource(path, ln, cl, "invalid NUL character"); found = true; + } else { if (c >= 128u8 && width == 1) { + pkgfailsource(path, ln, cl, "invalid UTF-8 encoding"); + found = true; }; - if (src[i] == '\n') { + }; + if (c == '\n') { ln += 1; cl = 1; } else { - cl += 1; + cl += width; }; - i += 1; + i += width; }; return found; }; @@ -2299,7 +2310,7 @@ export fn packagecommand(args: []str) int = { pkgfailpath(ds.paths[i], "invalid or missing package clause"); return pkgteststatusfail(explicitstatus, compileonly); }; - if (pkgnulerrors(ds.paths[i], body)) { + if (pkgsourceerrors(ds.paths[i], body)) { return pkgteststatusfail(explicitstatus, compileonly); }; if (!pkgclause(body, &pn)) { diff --git a/lib/ww/syntax/lex.ww b/lib/ww/syntax/lex.ww index b424d3dc..5d645df2 100644 --- a/lib/ww/syntax/lex.ww +++ b/lib/ww/syntax/lex.ww @@ -52,6 +52,7 @@ export type lex = struct { col: i32, errs: i32, nulcount: u64, + utf8count: u64, // a `//ww:module-reset` directive was seen in the last skipped run; // lexnext emits TK_MODRESET before the next real token (#16 opt-B). modreset: i32, @@ -74,6 +75,54 @@ fn bomat(src: *u8, len: u64, off: u64) bool = { && src[i + 2] == 0xbfu8; }; +// Width of the valid UTF-8 scalar sequence beginning at off, or zero. This +// mirrors unicode/utf8.DecodeRune's validity partition without changing WW's +// established byte-oriented treatment of valid non-ASCII token text. +fn utf8seqwidth(src: *u8, len: u64, off: u64) i32 = { + if (off >= len) { return 0; }; + let c: u8 = src[off]; + if (c < 128u8) { return 1; }; + if (c >= 194u8 && c <= 223u8 && len - off >= 2u64 + && src[off + 1u64] >= 128u8 && src[off + 1u64] <= 191u8) { + return 2; + }; + if (len - off >= 3u64 && src[off + 2u64] >= 128u8 + && src[off + 2u64] <= 191u8) { + let c1: u8 = src[off + 1u64]; + if ((c == 224u8 && c1 >= 160u8 && c1 <= 191u8) + || (c >= 225u8 && c <= 236u8 && c1 >= 128u8 && c1 <= 191u8) + || (c == 237u8 && c1 >= 128u8 && c1 <= 159u8) + || (c >= 238u8 && c <= 239u8 && c1 >= 128u8 && c1 <= 191u8)) { + return 3; + }; + }; + if (len - off >= 4u64 && src[off + 2u64] >= 128u8 + && src[off + 2u64] <= 191u8 && src[off + 3u64] >= 128u8 + && src[off + 3u64] <= 191u8) { + let c1: u8 = src[off + 1u64]; + if ((c == 240u8 && c1 >= 144u8 && c1 <= 191u8) + || (c >= 241u8 && c <= 243u8 && c1 >= 128u8 && c1 <= 191u8) + || (c == 244u8 && c1 >= 128u8 && c1 <= 143u8)) { + return 4; + }; + }; + return 0; +}; + +fn utf8bytevalid(src: *u8, len: u64, off: u64) bool = { + if (utf8seqwidth(src, len, off) != 0) { return true; }; + let c: u8 = src[off]; + if (c < 128u8 || c > 191u8) { return false; }; + let back: u64 = 1u64; + for (back <= 3u64 && back <= off) { + if (utf8seqwidth(src, len, off - back) > (back: i32)) { + return true; + }; + back += 1u64; + }; + return false; +}; + export fn lexinit(l: *lex, file: str, src: *u8, len: u64) void = { l.file = file; l.src = src; @@ -83,6 +132,7 @@ export fn lexinit(l: *lex, file: str, src: *u8, len: u64) void = { l.col = 1; l.errs = 0; l.nulcount = 0u64; + l.utf8count = 0u64; l.modreset = 0; l.modpathset = 0; l.modresetpathset = 0; @@ -111,15 +161,34 @@ fn lskipnul(l: *lex) void = { }; }; -// Return the raw offset of a logical byte lookahead. Raw NUL bytes do not -// occupy a slot in the token stream: Go's source.nextch diagnoses them and -// immediately resumes decoding at the following character. +// Go 1.26.5 syntax.source.nextch diagnoses and discards one raw byte for each +// utf8.DecodeRune RuneError of width one. +fn lskiputf8(l: *lex) void = { + for (l.lpos < l.srclen + && srcb(l, l.lpos) >= 128 + && !utf8bytevalid(l.src, l.srclen, l.lpos)) { + let ep: pos; + ep.file = l.file; + ep.line = l.line; + ep.col = l.col; + l.lpos += 1u64; + l.col += 1; + errat(l, &ep, "invalid UTF-8 encoding"); + l.utf8count += 1u64; + }; +}; + +// Return the raw offset of a logical byte lookahead. Raw NUL and malformed +// UTF-8 bytes do not occupy a slot in the token stream: Go's source.nextch +// diagnoses them and immediately resumes at the following byte. fn lrawoff(l: *lex, ahead0: u64) u64 = { let p: u64 = l.lpos; let ahead: u64 = ahead0; for (true) { for (p < l.srclen) { - if (srcb(l, p) != 0) { break; }; + let b: i32 = srcb(l, p); + if (b != 0 && (b < 128 + || utf8bytevalid(l.src, l.srclen, p))) { break; }; p += 1u64; }; if (ahead == 0u64 || p >= l.srclen) { return p; }; @@ -135,6 +204,10 @@ fn lpeek(l: *lex, ahead: u64) i32 = { if (l.lpos >= l.srclen) { return -1; }; let c: i32 = srcb(l, l.lpos); if (c == 0) { lskipnul(l); continue; }; + if (c >= 128 && !utf8bytevalid(l.src, l.srclen, l.lpos)) { + lskiputf8(l); + continue; + }; if (ahead == 0u64) { if (bomat(l.src, l.srclen, l.lpos)) { return 0xFEFF; }; return c; @@ -151,6 +224,10 @@ fn lget(l: *lex) i32 = { if (l.lpos >= l.srclen) { return -1; }; let c: i32 = srcb(l, l.lpos); if (c == 0) { lskipnul(l); continue; }; + if (c >= 128 && !utf8bytevalid(l.src, l.srclen, l.lpos)) { + lskiputf8(l); + continue; + }; if (bomat(l.src, l.srclen, l.lpos)) { let bp: pos; bp.file = l.file; @@ -172,16 +249,17 @@ fn lget(l: *lex) i32 = { }; }; -// Copy one raw source span into token text while omitting diagnosed NUL -// bytes. This keeps keyword, numeric, suffix, and directive recovery on the -// same logical character stream as lpeek/lget. +// Copy one raw source span into token text while omitting diagnosed NUL and +// malformed UTF-8 bytes. This keeps keyword, numeric, suffix, and directive +// recovery on the same logical character stream as lpeek/lget. fn lexspan(l: *lex, begin: u64, end: u64) str = { let buf: []u8 = alloc([], end - begin + 1u64)!; let i: u64 = begin; let j: u64 = 0u64; for (i < end) { let b: i32 = srcb(l, i); - if (b != 0) { + if (b != 0 && (b < 128 + || utf8bytevalid(l.src, l.srclen, i))) { buf[j] = b: u8; j += 1u64; }; @@ -221,6 +299,7 @@ fn errat(l: *lex, p: *pos, msg: str) void = { fn linecomment(l: *lex) void = { let begin: u64 = l.lpos; let nulbegin: u64 = l.nulcount; + let utf8begin: u64 = l.utf8count; for (true) { let c: i32 = lpeek(l, 0u64); if (c < 0 || c == '\n') { break; }; @@ -228,7 +307,7 @@ fn linecomment(l: *lex) void = { }; let end: u64 = l.lpos; let body: str; - if (l.nulcount == nulbegin) { + if (l.nulcount == nulbegin && l.utf8count == utf8begin) { body.ptr = l.src + begin; body.len = (end - begin): i32; } else { @@ -534,6 +613,7 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { out.col = start.col; let begin: u64 = l.lpos; let nulbegin: u64 = l.nulcount; + let utf8begin: u64 = l.utf8count; let base: i32 = 10; let isfloat: bool = false; @@ -583,7 +663,7 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { }; let rawend: u64 = l.lpos; - if (l.nulcount == nulbegin) { + if (l.nulcount == nulbegin && l.utf8count == utf8begin) { let view: str; view.ptr = l.src + begin; view.len = (rawend - begin): i32; @@ -671,6 +751,7 @@ fn lexnum(l: *lex, start: *pos, out: *tok) void = { fn lexident(l: *lex, start: *pos, out: *tok) void = { let begin: u64 = l.lpos; let nulbegin: u64 = l.nulcount; + let utf8begin: u64 = l.utf8count; for (true) { let c: i32 = lpeek(l, 0u64); if (c < 0) { break; }; @@ -678,7 +759,7 @@ fn lexident(l: *lex, start: *pos, out: *tok) void = { lget(l); }; let text: str; - if (l.nulcount == nulbegin) { + if (l.nulcount == nulbegin && l.utf8count == utf8begin) { let view: str; view.ptr = l.src + begin; view.len = (l.lpos - begin): i32; diff --git a/lib/ww/syntax/tok_test.ww b/lib/ww/syntax/tok_test.ww index 7be27e06..21a982f4 100644 --- a/lib/ww/syntax/tok_test.ww +++ b/lib/ww/syntax/tok_test.ww @@ -437,3 +437,138 @@ fn doublebomerror() bool = { assert(bomerror("\"", "\"")); assert(bomerror("'", "'")); }; + +fn utf8source(dst: *u8, before: str, middle: *u8, middlen: i32, + after: str) u64 = { + let n: i32 = 0; + let i: i32 = 0; + for (i < before.len) { dst[n] = before[i]; n += 1; i += 1; }; + i = 0; + for (i < middlen) { dst[n] = middle[i]; n += 1; i += 1; }; + i = 0; + for (i < after.len) { dst[n] = after[i]; n += 1; i += 1; }; + return n: u64; +}; + +fn checkmalformedutf8(raw: *u8, rawlen: i32, errs: i32) void = { + let src: [32]u8; + let n: u64 = utf8source(src.ptr, "f", raw, rawlen, "n bar"); + let l: syntax.lex; + syntax.lexinit(&l, "t", src.ptr, n); + let t: syntax.tok; + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN && t.line == 1 && t.col == 1); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_IDENT && t.text == "bar"); + assert(t.line == 1 && t.col == 4 + rawlen); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_EOF && l.errs == errs); +}; + +// Go 1.26.5 source.nextch reports and filters one byte for every +// DecodeRune width-one malformed result. +@test fn malformed_utf8_width_one_recovery() void = { + let lead: [1]u8 = [0xffu8]; + let continuation: [1]u8 = [0x80u8]; + let overlong: [2]u8 = [0xc0u8, 0x80u8]; + let surrogate: [3]u8 = [0xedu8, 0xa0u8, 0x80u8]; + let outofrange: [4]u8 = [0xf4u8, 0x90u8, 0x80u8, 0x80u8]; + let truncated: [2]u8 = [0xe2u8, 0x82u8]; + checkmalformedutf8(lead.ptr, 1, 1); + checkmalformedutf8(continuation.ptr, 1, 1); + checkmalformedutf8(overlong.ptr, 2, 2); + checkmalformedutf8(surrogate.ptr, 3, 3); + checkmalformedutf8(outofrange.ptr, 4, 4); + checkmalformedutf8(truncated.ptr, 2, 2); +}; + +@test fn valid_utf8_is_preserved() void = { + let raw: [12]u8 = [ + 0xc3u8, 0xa9u8, + 0xeau8, 0xb0u8, 0x80u8, + 0xefu8, 0xbfu8, 0xbdu8, + 0xf0u8, 0x9fu8, 0x98u8, 0x80u8, + ]; + let src: [32]u8; + let n: u64 = utf8source(src.ptr, "\"", raw.ptr, 12, "\" fn"); + let l: syntax.lex; + syntax.lexinit(&l, "t", src.ptr, n); + let t: syntax.tok; + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_STR && t.text.len == 12); + let i: i32 = 0; + for (i < 12) { assert(t.text[i] == raw[i]); i += 1; }; + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN && t.line == 1 && t.col == 16); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_EOF && l.errs == 0); + + n = utf8source(src.ptr, "// ", raw.ptr, 12, "\nfn"); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN && t.line == 2 && t.col == 1); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_EOF && l.errs == 0); +}; + +@test fn malformed_utf8_token_recovery() void = { + let bad: [1]u8 = [0xffu8]; + let src: [64]u8; + let l: syntax.lex; + let t: syntax.tok; + let n: u64 = utf8source(src.ptr, "1", bad.ptr, 1, "_0"); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_INT && t.uval == 10u64 && l.errs == 1); + + n = utf8source(src.ptr, "=", bad.ptr, 1, "="); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_EQ && l.errs == 1); + + n = utf8source(src.ptr, "/", bad.ptr, 1, "/ comment\nfn"); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN && l.errs == 1); + + n = utf8source(src.ptr, "/* end *", bad.ptr, 1, "/ fn"); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN && l.errs == 1); + + n = utf8source(src.ptr, "\"a", bad.ptr, 1, "b\""); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_STR && t.text == "ab" && l.errs == 1); + + n = utf8source(src.ptr, "\"\\", bad.ptr, 1, "n\""); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_STR && t.text.len == 1); + assert(t.text[0] == '\n' && l.errs == 1); + + n = utf8source(src.ptr, "\"\\x0", bad.ptr, 1, "0\""); + syntax.lexinit(&l, "t", src.ptr, n); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_STR && t.text.len == 1); + assert(t.text[0] == 0u8 && l.errs == 1); +}; + +@test fn malformed_utf8_bom_nul_columns() void = { + let src: [12]u8 = [ + 0xefu8, 0xbbu8, 0xbfu8, + 'f': u8, 0xffu8, 0u8, 'n': u8, + ' ': u8, 'b': u8, 'a': u8, 'r': u8, 0u8, + ]; + let l: syntax.lex; + syntax.lexinit(&l, "t", src.ptr, 11u64); + let t: syntax.tok; + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_FN && t.line == 1 && t.col == 4); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_IDENT && t.text == "bar"); + assert(t.line == 1 && t.col == 9); + syntax.lexnext(&l, &t); + assert(t.kind == syntax.tkind.TK_EOF); + assert(l.errs == 2 && l.nulcount == 1u64); +}; diff --git a/test/package/package_test.ww b/test/package/package_test.ww index dc38af71..a2c62c5c 100644 --- a/test/package/package_test.ww +++ b/test/package/package_test.ww @@ -161,6 +161,38 @@ fn rewritenulfile(path: str, before: str, after: str) void = { putnulfile(path, before, after, true); }; +fn putinvalidutf8file(path: str, before: str, after: str, + rewrite: bool) void = { + let flags: os.flag = os.flag.WRONLY; + if (rewrite) { flags |= os.flag.TRUNC; } + else { flags |= os.flag.CREATE | os.flag.EXCL; }; + let fd: i32 = os.open(path, flags, 384i32); + assert(fd >= 0); + match (os.writeall(fd, before.ptr, before.len: u64)) { + case let n: i64 => assert(n == before.len: i64); + case let e: os.oserror => abort("write before invalid UTF-8 failed"); + }; + let invalid: [1]u8; + invalid[0] = 0xffu8; + match (os.writeall(fd, invalid.ptr, 1u64)) { + case let n: i64 => assert(n == 1i64); + case let e: os.oserror => abort("write invalid UTF-8 failed"); + }; + match (os.writeall(fd, after.ptr, after.len: u64)) { + case let n: i64 => assert(n == after.len: i64); + case let e: os.oserror => abort("write after invalid UTF-8 failed"); + }; + assert(os.close(fd) == 0); +}; + +fn writeinvalidutf8file(path: str, before: str, after: str) void = { + putinvalidutf8file(path, before, after, false); +}; + +fn rewriteinvalidutf8file(path: str, before: str, after: str) void = { + putinvalidutf8file(path, before, after, true); +}; + fn writedoublenulfile(path: str, before: str, after: str) void = { let fd: i32 = os.open(path, os.flag.WRONLY | os.flag.CREATE | os.flag.EXCL, 384i32); @@ -17925,3 +17957,382 @@ fn runtimepath(relative: str) str = { && !directoryhasnew(root)); clean(root); }; + +// Go 1.26.5 removes each malformed UTF-8 byte from the logical source after +// reporting it. WW applies that decoder rule only after physical-file +// eligibility and before package clauses or imports can affect the graph. +@test fn malformed_utf8_is_rejected_in_every_selected_source() void = { + let root: str = fresh(); + let source: str = strings.concat(root, "/source"); + mkdirall(source); + + // Filtering the invalid byte reconstructs the keyword, so the decoder's + // positioned error is the only failure and neither frontend emits assembly. + let invaliddirect: str = strings.concat(root, "/invalid-direct.ww"); + let validdirect: str = strings.concat(root, "/valid-direct.ww"); + writeinvalidutf8file(invaliddirect, "pack", + "age main;\nfn main() i32 = { return 0; };\n"); + writefile(validdirect, strings.concat( + "package main;\n", + "// 한국어, café, and � are valid UTF-8 source bytes.\n", + "fn main() i32 = { return 0; };\n")); + let compilers: []str = ["w6c", "w6c_ww"]; + let stages: []str = ["ww", "ww_ww"]; + let tags: []str = ["c", "ww"]; + let invaliddiags: []str = ["", ""]; + let validasms: []str = [strings.concat(root, "/valid-c.s"), + strings.concat(root, "/valid-ww.s")]; + let si: i32 = 0; + for (si < compilers.len) { + let invalidasm: str = strings.concat(root, "/invalid-", tags[si], ".s"); + let invalidav: []str = [driver(compilers[si]), "-c", + "--command-package", "-o", invalidasm, invaliddirect]; + let out: commandout; + runcommand(root, strings.concat("utf8-direct-invalid-", tags[si]), + invalidav, (30i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, ":1:5: error: invalid UTF-8 encoding\n") + && occurrences(out.stderr, "invalid UTF-8 encoding") == 1 + && !has(out.stderr, "unexpected character") + && !has(out.stderr, "invalid or missing package clause") + && !os.exists(invalidasm)); + invaliddiags[si] = strings.dup(out.stderr); + + let validav: []str = [driver(compilers[si]), "-c", + "--command-package", "-o", validasms[si], validdirect]; + runcommand(root, strings.concat("utf8-direct-valid-", tags[si]), + validav, (30i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0); + si += 1; + }; + assert(same(invaliddiags[0], invaliddiags[1])); + assert(same(readfile(validasms[0]), readfile(validasms[1]))); + + let dep: str = strings.concat(source, "/dep"); + let app: str = strings.concat(source, "/app"); + let bad: str = strings.concat(source, "/bad"); + let good: str = strings.concat(source, "/good"); + let ordered: str = strings.concat(source, "/ordered"); + mkdirall(dep); mkdirall(app); mkdirall(bad); mkdirall(good); + mkdirall(ordered); + writeinvalidutf8file(strings.concat(dep, "/dep.ww"), + "package dep;\n// bad ", + " dependency\nexport fn value() i32 = { return 0; };\n"); + writefile(strings.concat(app, "/main.ww"), strings.concat( + "package main;\nimport dep;\n", + "fn main() i32 = { return dep.value(); };\n")); + writeinvalidutf8file(strings.concat(bad, "/main.ww"), + "package main;\n// bad ", strings.concat( + " root\nimport missing;\n", + "fn main() i32 = { return 0; };\n")); + writefile(strings.concat(good, "/main.ww"), + "package main;\nfn main() i32 = { return 0; };\n"); + writeinvalidutf8file(strings.concat(good, "/ignored_windows.ww"), + "not package ", " and not valid WW\n"); + writeinvalidutf8file(strings.concat(good, "/ignored_test.ww"), + "not package ", " and not valid WW\n"); + writeinvalidutf8file(strings.concat(ordered, "/z.ww"), + "package main;\n// z ", "\nfn z() void = {};\n"); + writefile(strings.concat(ordered, "/b.ww"), + "package main;\nfn main() i32 = { return 0; };\n"); + writeinvalidutf8file(strings.concat(ordered, "/a.ww"), + "package main;\n// a ", "\nfn a() void = {};\n"); + + // Root and dependency decoding precede graph resolution and all producers. + // Excluded files are not semantic inputs, and selection remains byte-sorted. + let goodbins: []str = [strings.concat(root, "/good-c"), + strings.concat(root, "/good-ww")]; + let depdiags: []str = ["", ""]; + let rootdiags: []str = ["", ""]; + let orderdiags: []str = ["", ""]; + si = 0; + for (si < stages.len) { + let out: commandout; + let depwork: str = strings.concat(root, "/dep-work-", tags[si]); + let depout: str = strings.concat(root, "/dep-output-", tags[si]); + mkdirall(depwork); + let depav: []str = [driver(stages[si]), "build", "-w", depwork, + "-I", source, "-o", depout, "app"]; + runcommand(root, strings.concat("utf8-dependency-", tags[si]), depav, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, + "/dep/dep.ww:2:8: error: invalid UTF-8 encoding\n") + && occurrences(out.stderr, "invalid UTF-8 encoding") == 1 + && !has(out.stderr, "w6c failed") + && !os.exists(depout) && directoryisempty(depwork)); + depdiags[si] = strings.dup(out.stderr); + + let badwork: str = strings.concat(root, "/bad-work-", tags[si]); + let badout: str = strings.concat(root, "/bad-output-", tags[si]); + mkdirall(badwork); + let badav: []str = [driver(stages[si]), "build", "-w", badwork, + "-I", source, "-o", badout, "bad"]; + runcommand(root, strings.concat("utf8-root-", tags[si]), badav, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, + "/bad/main.ww:2:8: error: invalid UTF-8 encoding\n") + && occurrences(out.stderr, "invalid UTF-8 encoding") == 1 + && !has(out.stderr, "cannot find import") + && !os.exists(badout) && directoryisempty(badwork)); + rootdiags[si] = strings.dup(out.stderr); + + let orderwork: str = strings.concat(root, "/order-work-", tags[si]); + let orderout: str = strings.concat(root, "/order-output-", tags[si]); + mkdirall(orderwork); + let orderav: []str = [driver(stages[si]), "build", "-w", orderwork, + "-I", source, "-o", orderout, "ordered"]; + runcommand(root, strings.concat("utf8-order-", tags[si]), orderav, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, + "/ordered/a.ww:2:6: error: invalid UTF-8 encoding\n") + && !has(out.stderr, "/ordered/z.ww:") + && !os.exists(orderout) && directoryisempty(orderwork)); + orderdiags[si] = strings.dup(out.stderr); + + let goodwork: str = strings.concat(root, "/good-work-", tags[si]); + mkdirall(goodwork); + let goodav: []str = [driver(stages[si]), "build", "-w", goodwork, + "-I", source, "-o", goodbins[si], "good"]; + runcommand(root, strings.concat("utf8-excluded-", tags[si]), goodav, + (120i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0 + && !directoryhasnew(goodwork) + && !directoryhasfragment(goodwork, ".wwtxn.")); + let runav: []str = [goodbins[si]]; + runcommand(root, strings.concat("utf8-good-run-", tags[si]), runav, + time.second, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0); + si += 1; + }; + assert(same(depdiags[0], depdiags[1]) + && same(rootdiags[0], rootdiags[1]) + && same(orderdiags[0], orderdiags[1]) + && same(readfile(goodbins[0]), readfile(goodbins[1]))); + + // Every selected test-source role rejects before generated-main creation, + // execution, accounting, and retained-binary publication. + let prodcase: str = strings.concat(source, "/prodcase"); + let samecase: str = strings.concat(source, "/samecase"); + let externalcase: str = strings.concat(source, "/externalcase"); + let onlycase: str = strings.concat(source, "/onlycase"); + mkdirall(prodcase); mkdirall(samecase); mkdirall(externalcase); + mkdirall(onlycase); + writeinvalidutf8file(strings.concat(prodcase, "/prod.ww"), + "package prodcase;\n// bad ", + " production\nfn value() i32 = { return 1; };\n"); + writefile(strings.concat(prodcase, "/prod_test.ww"), + "package prodcase;\n@test fn must_not_run() void = { abort(); };\n"); + writefile(strings.concat(samecase, "/prod.ww"), + "package samecase;\nfn value() i32 = { return 1; };\n"); + writeinvalidutf8file(strings.concat(samecase, "/same_test.ww"), + "package samecase;\n// bad ", + " same\n@test fn must_not_run() void = { abort(); };\n"); + writefile(strings.concat(externalcase, "/prod.ww"), + "package externalcase;\nexport fn value() i32 = { return 1; };\n"); + writeinvalidutf8file(strings.concat(externalcase, "/external_test.ww"), + "package externalcase_test;\n// bad ", strings.concat( + " external\nimport externalcase;\n", + "@test fn must_not_run() void = { abort(); };\n")); + writeinvalidutf8file(strings.concat(onlycase, "/only_test.ww"), "", + "package onlycase;\n@test fn must_not_run() void = { abort(); };\n"); + let testids: []str = ["prodcase", "samecase", "externalcase", "onlycase"]; + let testpositions: []str = [ + "/prodcase/prod.ww:2:8: error: invalid UTF-8 encoding\n", + "/samecase/same_test.ww:2:8: error: invalid UTF-8 encoding\n", + "/externalcase/external_test.ww:2:8: error: invalid UTF-8 encoding\n", + "/onlycase/only_test.ww:1:1: error: invalid UTF-8 encoding\n"]; + let testdiags: []str = ["", "", "", ""]; + si = 0; + for (si < stages.len) { + let ti: i32 = 0; + for (ti < testids.len) { + let testwork: str = strings.concat(root, "/test-work-", tags[si], + "-", boundarypkgname(ti)); + let testout: str = strings.concat(root, "/test-output-", tags[si], + "-", boundarypkgname(ti)); + mkdirall(testwork); + let av: []str = [driver(stages[si]), "test", "-w", testwork, + "-I", source, "-o", testout, testids[ti]]; + let out: commandout; + runcommand(root, strings.concat("utf8-test-", tags[si], "-", + boundarypkgname(ti)), av, + (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(same(out.stdout, "FAIL\n") + && has(out.stderr, testpositions[ti]) + && occurrences(out.stderr, "invalid UTF-8 encoding") == 1 + && !has(out.stdout, "must_not_run") + && !has(out.stdout, "discovered") + && !has(out.stdout, "passed") && !has(out.stdout, "failed") + && !has(out.stdout, "ok ") + && !os.exists(testout) && directoryisempty(testwork)); + if (si == 0) { testdiags[ti] = strings.dup(out.stderr); } + else { assert(same(testdiags[ti], out.stderr)); }; + ti += 1; + }; + si += 1; + }; + + // Invalid warm input reaches no tool and preserves every committed action, + // tool record, and public byte. Exact source restoration is a cache hit. + let warm: str = strings.concat(source, "/warm"); + mkdirall(warm); + let warmpath: str = strings.concat(warm, "/main.ww"); + let warmvalid: str = + "package main;\nfn main() i32 = { return 0; };\n"; + writefile(warmpath, warmvalid); + let wrapper: str = strings.concat(root, "/utf8-compiler-wrapper.sh"); + writeexecutable(wrapper, strings.concat( + "#!/bin/sh\n", + "printf 'compile\\n' >> \"$WW_UTF8_TRACE\"\n", + "exec \"$WW_UTF8_REAL\" \"$@\"\n")); + let commandsuffixes: []str = [".unit.ww", ".wwi", ".s", ".o", ".a", + ".init.unit.ww", ".init.s", ".init.o"]; + let toolpaths: []str = ["/.wwtool.ww", "/.wwtool.w6c", + "/.wwtool.w6a", "/.wwtool.stamp"]; + si = 0; + for (si < stages.len) { + let warmwork: str = strings.concat(root, "/warm-work-", tags[si]); + let warmout: str = strings.concat(root, "/warm-output-", tags[si]); + let trace: str = strings.concat(root, "/warm-trace-", tags[si]); + mkdirall(warmwork); writefile(trace, ""); + let env: []str = os.getenvs(); + append(env, strings.concat("WW_W6C=", wrapper)); + append(env, strings.concat("WW_UTF8_TRACE=", trace)); + append(env, strings.concat("WW_UTF8_REAL=", driver(compilers[si]))); + let av: []str = [driver(stages[si]), "build", "-w", warmwork, + "-I", source, "-o", warmout, "warm"]; + let out: commandout; + runcommandenv(root, strings.concat("utf8-warm-cold-", tags[si]), av, + env, (120i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0 + && readfile(trace).len != 0); + let refs: []str = alloc([], commandsuffixes.len: u64)!; + let fi: i32 = 0; + for (fi < commandsuffixes.len) { + append(refs, strings.dup(readfile(strings.concat(warmwork, + "/warm", commandsuffixes[fi])))); + fi += 1; + }; + let toolrefs: []str = alloc([], toolpaths.len: u64)!; + fi = 0; + for (fi < toolpaths.len) { + append(toolrefs, strings.dup(readfile(strings.concat(warmwork, + toolpaths[fi])))); + fi += 1; + }; + let binref: str = strings.dup(readfile(warmout)); + rewritefile(trace, ""); + rewriteinvalidutf8file(warmpath, "package main;\n// bad ", + " warm\nfn main() i32 = { return 0; };\n"); + runcommandenv(root, strings.concat("utf8-warm-invalid-", tags[si]), av, + env, (60i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 1); + assert(out.stdout.len == 0 + && has(out.stderr, ":2:8: error: invalid UTF-8 encoding\n") + && readfile(trace).len == 0 && same(binref, readfile(warmout))); + fi = 0; + for (fi < commandsuffixes.len) { + assert(same(refs[fi], readfile(strings.concat(warmwork, + "/warm", commandsuffixes[fi])))); + fi += 1; + }; + fi = 0; + for (fi < toolpaths.len) { + assert(same(toolrefs[fi], readfile(strings.concat(warmwork, + toolpaths[fi])))); + fi += 1; + }; + assert(!directoryhasnew(warmwork) + && !directoryhasfragment(warmwork, ".wwtxn.")); + rewritefile(warmpath, warmvalid); + runcommandenv(root, strings.concat("utf8-warm-restored-", tags[si]), av, + env, (120i64 * (time.second: i64)): time.duration, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0 + && readfile(trace).len == 0 && same(binref, readfile(warmout))); + fi = 0; + for (fi < commandsuffixes.len) { + assert(same(refs[fi], readfile(strings.concat(warmwork, + "/warm", commandsuffixes[fi])))); + fi += 1; + }; + fi = 0; + for (fi < toolpaths.len) { + assert(same(toolrefs[fi], readfile(strings.concat(warmwork, + toolpaths[fi])))); + fi += 1; + }; + si += 1; + }; + + // Validation is request-local: an overlapping valid Cstage request is not + // cancelled or contaminated by an invalid WWstage request. + let pcwork: str = strings.concat(root, "/parallel-c-work"); + let pwwork: str = strings.concat(root, "/parallel-ww-work"); + let pcout: str = strings.concat(root, "/parallel-c-output"); + let pwout: str = strings.concat(root, "/parallel-ww-output"); + mkdirall(pcwork); mkdirall(pwwork); + let pcav: []str = [driver("ww"), "build", "-w", pcwork, + "-I", source, "-o", pcout, "good"]; + let pwav: []str = [driver("ww_ww"), "build", "-w", pwwork, + "-I", source, "-o", pwout, "bad"]; + let pc: exec.command; + pc.path = pcav[0]; pc.argv = pcav; pc.env = os.getenvs(); pc.dir = repo(); + pc.stdoutpath = strings.concat(root, "/parallel-c.stdout"); + pc.stderrpath = strings.concat(root, "/parallel-c.stderr"); + pc.deadline = time.add(time.now(time.clock.monotonic), + (120i64 * (time.second: i64)): time.duration); + pc.grace = (100i64 * (time.millisecond: i64)): time.duration; + let pw: exec.command; + pw.path = pwav[0]; pw.argv = pwav; pw.env = os.getenvs(); pw.dir = repo(); + pw.stdoutpath = strings.concat(root, "/parallel-ww.stdout"); + pw.stderrpath = strings.concat(root, "/parallel-ww.stderr"); + pw.deadline = time.add(time.now(time.clock.monotonic), + (120i64 * (time.second: i64)): time.duration); + pw.grace = (100i64 * (time.millisecond: i64)): time.duration; + let pcp: exec.process; + let pwp: exec.process; + exec.start(&pcp, &pc); exec.start(&pwp, &pw); + let pcdone: bool = false; + let pwdone: bool = false; + for (!pcdone || !pwdone) { + if (!pcdone) { pcdone = exec.poll(&pcp); }; + if (!pwdone) { pwdone = exec.poll(&pwp); }; + if (!pcdone || !pwdone) { + time.sleep(time.millisecond, time.clock.monotonic); + }; + }; + assert(pcp.result.errno == 0 && pcp.result.cleanuperrno == 0 + && pcp.result.termination == exec.termination.EXIT + && pcp.result.code == 0); + assert(pwp.result.errno == 0 && pwp.result.cleanuperrno == 0 + && pwp.result.termination == exec.termination.EXIT + && pwp.result.code == 1); + assert(readfile(pc.stdoutpath).len == 0 && readfile(pc.stderrpath).len == 0); + assert(readfile(pw.stdoutpath).len == 0 + && has(readfile(pw.stderrpath), "invalid UTF-8 encoding")); + assert(same(readfile(pcout), readfile(goodbins[0])) && !os.exists(pwout)); + assert(directoryisempty(pwwork) && !directoryhasnew(pcwork) + && !directoryhasfragment(pcwork, ".wwtxn.")); + let runav: []str = [pcout]; + let out: commandout; + runcommand(root, "utf8-parallel-run", runav, time.second, &out); + expectexit(&out, 0); + assert(out.stdout.len == 0 && out.stderr.len == 0); + assert(!directoryhasfragment(root, ".wwtxn.") + && !directoryhasfragment(root, ".install") + && !directoryhasnew(root)); + clean(root); +}; diff --git a/test/wcc/100_lex.c b/test/wcc/100_lex.c index 3a4990d5..b2b33f44 100644 --- a/test/wcc/100_lex.c +++ b/test/wcc/100_lex.c @@ -240,6 +240,113 @@ static const struct { { "f\0\0n", sizeof "f\0\0n" - 1, "fn", 2 }, }; +/* Go 1.26.5 source.nextch reports each DecodeRune width-one result and + * removes that byte before scanner recovery. */ +static const struct { + const char *src; + size_t len; + const char *expect; + int errs; +} utf8rows[] = { + { "f" "\xff" "n bar", sizeof "f" "\xff" "n bar" - 1, + "fn IDENT(bar)", 1 }, + { "f" "\x80" "n bar", sizeof "f" "\x80" "n bar" - 1, + "fn IDENT(bar)", 1 }, + { "f" "\xc0\x80" "n bar", sizeof "f" "\xc0\x80" "n bar" - 1, + "fn IDENT(bar)", 2 }, + { "f" "\xed\xa0\x80" "n bar", + sizeof "f" "\xed\xa0\x80" "n bar" - 1, + "fn IDENT(bar)", 3 }, + { "f" "\xf4\x90\x80\x80" "n bar", + sizeof "f" "\xf4\x90\x80\x80" "n bar" - 1, + "fn IDENT(bar)", 4 }, + { "f" "\xe2\x82" "n bar", sizeof "f" "\xe2\x82" "n bar" - 1, + "fn IDENT(bar)", 2 }, + { "1" "\xff" "_0", sizeof "1" "\xff" "_0" - 1, + "INT(10)", 1 }, + { "=" "\xff" "=", sizeof "=" "\xff" "=" - 1, + "==", 1 }, + { "/" "\xff" "/ comment\nfn", + sizeof "/" "\xff" "/ comment\nfn" - 1, "fn", 1 }, + { "/* end *" "\xff" "/ fn", + sizeof "/* end *" "\xff" "/ fn" - 1, "fn", 1 }, + { "\"a" "\xff" "b\"", sizeof "\"a" "\xff" "b\"" - 1, + "STR(ab)", 1 }, +}; + +static int +runutf8valid(void) +{ + static const char strsrc[] = + "\"" "\xc3\xa9" "\xea\xb0\x80" "\xef\xbf\xbd" + "\xf0\x9f\x98\x80" "\" fn"; + static const char commentsrc[] = + "// " "\xc3\xa9" "\xea\xb0\x80" "\xef\xbf\xbd" + "\xf0\x9f\x98\x80" "\nfn"; + static const unsigned char want[] = { + 0xc3, 0xa9, 0xea, 0xb0, 0x80, 0xef, 0xbf, 0xbd, + 0xf0, 0x9f, 0x98, 0x80, + }; + Arena *a = newarena(); + Lex l; + lexinit(&l, a, "", strsrc, sizeof strsrc - 1); + Tok s = lexnext(&l); + Tok f = lexnext(&l); + int ok = s.kind == TK_STR && s.tlen == sizeof want + && memcmp(s.text, want, sizeof want) == 0 + && f.kind == TK_FN && f.pos.col == 16 && l.errs == 0 + && lexnext(&l).kind == TK_EOF; + lexinit(&l, a, "", commentsrc, sizeof commentsrc - 1); + f = lexnext(&l); + ok = ok && f.kind == TK_FN && f.pos.line == 2 && f.pos.col == 1 + && l.errs == 0 && lexnext(&l).kind == TK_EOF; + if (!ok) + fprintf(stderr, "valid UTF-8 preservation failed\n"); + freearena(a); + return ok; +} + +static int +runutf8cols(void) +{ + static const char src[] = + "\xef\xbb\xbf" "f" "\xff" "\0" "n bar"; + Arena *a = newarena(); + Lex l; + lexinit(&l, a, "", src, sizeof src - 1); + Tok f = lexnext(&l); + Tok b = lexnext(&l); + int ok = f.kind == TK_FN && f.pos.line == 1 && f.pos.col == 4 + && b.kind == TK_IDENT && b.pos.line == 1 && b.pos.col == 9 + && strcmp(b.text, "bar") == 0 && l.errs == 2 + && l.nulcount == 1 && lexnext(&l).kind == TK_EOF; + if (!ok) + fprintf(stderr, "UTF-8 raw-byte columns or BOM/NUL recovery failed\n"); + freearena(a); + return ok; +} + +static int +runutf8escape(void) +{ + static const char line[] = "\"\\" "\xff" "n\""; + static const char hex[] = "\"\\x0" "\xff" "0\""; + Arena *a = newarena(); + Lex l; + lexinit(&l, a, "", line, sizeof line - 1); + Tok t = lexnext(&l); + int ok = t.kind == TK_STR && t.tlen == 1 && t.text[0] == '\n' + && l.errs == 1 && lexnext(&l).kind == TK_EOF; + lexinit(&l, a, "", hex, sizeof hex - 1); + t = lexnext(&l); + ok = ok && t.kind == TK_STR && t.tlen == 1 && t.text[0] == '\0' + && l.errs == 1 && lexnext(&l).kind == TK_EOF; + if (!ok) + fprintf(stderr, "UTF-8 escape recovery failed\n"); + freearena(a); + return ok; +} + static int runescapenul(void) { @@ -333,10 +440,23 @@ main(void) fail++; } } + for (size_t i = 0; i < sizeof utf8rows / sizeof utf8rows[0]; i++) { + if (!runrown(utf8rows[i].src, utf8rows[i].len, + utf8rows[i].expect, utf8rows[i].errs)) { + fprintf(stderr, "UTF-8 row %zu failed\n", i); + fail++; + } + } if (!runescapenul()) fail++; if (!runsuffixnul()) fail++; + if (!runutf8valid()) + fail++; + if (!runutf8cols()) + fail++; + if (!runutf8escape()) + fail++; if (!runlongdirective()) fail++; if (fail) {