w6c+wcc+selfhost+lib: int-cast truncate + use_alias, 5 new modules
Two cgen/check bugs surfaced by new lib modules, plus the modules
themselves (crc64, siphash, random, base64, base32).
1. `(big_u64): u32` (and `: u16`, `: u8`, `: bool`) didn't truncate.
N_CAST emitted nothing for int↔int; the value stayed in AX with
its upper bits intact and downstream CMPQ/DIVQ misread the slot.
The TK_TILDE path already had clamp logic for the same reason —
N_CAST was the missing case. Both stages now MOVL r,r for u32 and
ANDQ $mask for u8/u16/bool. Signed-narrow (i8/i16/i32) stays
no-op until w6a grows reg-reg MOVSBQ/MOVSWQ/MOVSXD. selfhost
cgcast walks alias chains via aliaslookup before checking
primsize/typenameisunsigned so `(u: random)` where
`type random = u64` still bypasses the clamp.
See cmd/w6c/cgen.c N_CAST and selfhost/cmd/wcc/cgenexpr.ww cgcast.
2. `mod.mod` type refs (`random.random` when the imported module
declares `export type random = u64;`) failed with "unknown type".
The driver concatenates imports into one flat scope, so SK_USE
`random` collided with SK_TYPE `random` and scope_define silently
dropped the use. resolve_typename's leaf lookup required
`kind == SK_USE` and gave up. Adds a `use_alias` flag to Sym; the
pass-1 decl scan now marks colliding syms in both directions
(use-after-type and type-after-use). resolve_typename and the
N_DOT cexpr branch treat `use_alias` like SK_USE for qualified
lookup. selfhost check.ww was already lenient on this path so no
ww-side change was needed; bootstrap fixed point (990-995) holds.
See cmd/wcc/check.c installdecl pass + N_DOT/resolve_typename and
cmd/wcc/ww.h Sym.use_alias.
New modules under lib/, each with @test vectors in *_test.ww and wired
into test/wcc/900_stdlib.c (26 modules → all compile):
- lib/hash/crc64 ECMA, ISO (mirror of crc32 shape)
- lib/hash/siphash SipHash-2-4, buffer-based sum/sum24
- lib/math/random SplitMix64 (init, next, u32n, u64n)
- lib/encoding/base64 RFC 4648 std + url-safe encode/decode + sizes
- lib/encoding/base32 RFC 4648 std + base32hex encode/decode + sizes
This commit is contained in:
182
lib/encoding/base64/base64.ww
Normal file
182
lib/encoding/base64/base64.ww
Normal file
@@ -0,0 +1,182 @@
|
||||
// encoding/base64 — RFC 4648 base64 encode/decode, buffer-based.
|
||||
//
|
||||
// Mirrors Hare's encoding::base64 surface, modulo Hare's stream-based
|
||||
// encoder/decoder. ww ships the in-memory subset only: `encode(dst,
|
||||
// src)` writes the encoded bytes into `dst`, returning the count;
|
||||
// `decode(dst, src)` writes the decoded bytes into `dst`, returning a
|
||||
// count or invalid.
|
||||
//
|
||||
// std uses '+' and '/' for indexes 62 and 63 (the RFC 4648 §4
|
||||
// alphabet); url uses '-' and '_' (the §5 url-safe alphabet). Both
|
||||
// pad encoded output with '=' to a multiple of 4 bytes.
|
||||
|
||||
// invalid — input was not well-formed base64 (bad char, wrong length,
|
||||
// padding error). Payload is the byte index of the first offending
|
||||
// position. Matches Hare's errors::invalid pairing with strconv.
|
||||
export type invalid = !i32;
|
||||
|
||||
// encodedsize — bytes required to encode `n` source bytes (including
|
||||
// '=' padding). Hare names it the same.
|
||||
export fn encodedsize(n: i32) i32 = {
|
||||
if (n == 0) { return 0; };
|
||||
return ((n - 1) / 3 + 1) * 4;
|
||||
};
|
||||
|
||||
// decodedsize — upper bound on the number of bytes decoded from `n`
|
||||
// encoded bytes. The exact count depends on padding; callers consult
|
||||
// the i32 returned by `decode`.
|
||||
export fn decodedsize(n: i32) i32 = {
|
||||
return (n / 4) * 3;
|
||||
};
|
||||
|
||||
// encchar — map a 6-bit value to its alphabet character. `urlsafe`
|
||||
// chooses '-'/'_' instead of '+'/'/' for 62/63.
|
||||
fn encchar(v: u8, urlsafe: bool) u8 = {
|
||||
if (v < 26u8) { return v + 65u8; }; // 'A' + v
|
||||
if (v < 52u8) { return v + 71u8; }; // 'a' + (v - 26) = v + 71
|
||||
if (v < 62u8) { return v - 4u8; }; // '0' + (v - 52) = v - 4
|
||||
if (v == 62u8) {
|
||||
if (urlsafe) { return 45u8; }; // '-'
|
||||
return 43u8; // '+'
|
||||
};
|
||||
if (urlsafe) { return 95u8; }; // '_'
|
||||
return 47u8; // '/'
|
||||
};
|
||||
|
||||
// decchar — inverse of encchar. Returns 0..63 on success or 255 on
|
||||
// invalid char. '=' is handled in the decode loop, not here.
|
||||
fn decchar(c: u8, urlsafe: bool) u8 = {
|
||||
if (c >= 65u8) { if (c <= 90u8) { return c - 65u8; }; }; // 'A'..'Z'
|
||||
if (c >= 97u8) { if (c <= 122u8) { return c - 71u8; }; }; // 'a'..'z'
|
||||
if (c >= 48u8) { if (c <= 57u8) { return c + 4u8; }; }; // '0'..'9'
|
||||
if (urlsafe) {
|
||||
if (c == 45u8) { return 62u8; }; // '-'
|
||||
if (c == 95u8) { return 63u8; }; // '_'
|
||||
} else {
|
||||
if (c == 43u8) { return 62u8; }; // '+'
|
||||
if (c == 47u8) { return 63u8; }; // '/'
|
||||
};
|
||||
return 255u8;
|
||||
};
|
||||
|
||||
// encodeinto — encode `src` into `dst` using the std (`urlsafe=false`)
|
||||
// or url-safe (`urlsafe=true`) alphabet. `dst` must hold at least
|
||||
// encodedsize(src.len) bytes. Returns the number of bytes written.
|
||||
fn encodeinto(dst: []u8, src: []u8, urlsafe: bool) i32 = {
|
||||
let i: i32 = 0;
|
||||
let j: i32 = 0;
|
||||
for (i + 2 < src.len) {
|
||||
let b0: u8 = src[i];
|
||||
let b1: u8 = src[i + 1];
|
||||
let b2: u8 = src[i + 2];
|
||||
dst[j] = encchar(b0 >> 2u8, urlsafe);
|
||||
dst[j + 1] = encchar(((b0 & 3u8) << 4u8) | (b1 >> 4u8), urlsafe);
|
||||
dst[j + 2] = encchar(((b1 & 15u8) << 2u8) | (b2 >> 6u8), urlsafe);
|
||||
dst[j + 3] = encchar(b2 & 63u8, urlsafe);
|
||||
i += 3;
|
||||
j += 4;
|
||||
};
|
||||
let rem: i32 = src.len - i;
|
||||
if (rem == 1) {
|
||||
let b0: u8 = src[i];
|
||||
dst[j] = encchar(b0 >> 2u8, urlsafe);
|
||||
dst[j + 1] = encchar((b0 & 3u8) << 4u8, urlsafe);
|
||||
dst[j + 2] = 61u8; // '='
|
||||
dst[j + 3] = 61u8; // '='
|
||||
j += 4;
|
||||
};
|
||||
if (rem == 2) {
|
||||
let b0: u8 = src[i];
|
||||
let b1: u8 = src[i + 1];
|
||||
dst[j] = encchar(b0 >> 2u8, urlsafe);
|
||||
dst[j + 1] = encchar(((b0 & 3u8) << 4u8) | (b1 >> 4u8), urlsafe);
|
||||
dst[j + 2] = encchar((b1 & 15u8) << 2u8, urlsafe);
|
||||
dst[j + 3] = 61u8; // '='
|
||||
j += 4;
|
||||
};
|
||||
return j;
|
||||
};
|
||||
|
||||
// encode — encode `src` into `dst` using the std alphabet. Returns
|
||||
// the number of bytes written. `dst` must hold at least
|
||||
// encodedsize(src.len) bytes.
|
||||
export fn encode(dst: []u8, src: []u8) i32 = {
|
||||
return encodeinto(dst, src, false);
|
||||
};
|
||||
|
||||
// encodeurl — same as encode but uses the url-safe alphabet ('-'/'_'
|
||||
// for 62/63).
|
||||
export fn encodeurl(dst: []u8, src: []u8) i32 = {
|
||||
return encodeinto(dst, src, true);
|
||||
};
|
||||
|
||||
// decodeinto — decode base64 `src` into `dst`. `dst` must hold at
|
||||
// least decodedsize(src.len) bytes. Returns the number of bytes
|
||||
// written, or invalid with the offending source index.
|
||||
fn decodeinto(dst: []u8, src: []u8, urlsafe: bool) (i32 | invalid) = {
|
||||
if (src.len == 0) { return 0; };
|
||||
if ((src.len & 3) != 0) { return src.len: invalid; };
|
||||
let i: i32 = 0;
|
||||
let j: i32 = 0;
|
||||
let end: i32 = src.len;
|
||||
for (i < end) {
|
||||
let c0: u8 = src[i];
|
||||
let c1: u8 = src[i + 1];
|
||||
let c2: u8 = src[i + 2];
|
||||
let c3: u8 = src[i + 3];
|
||||
let v0: u8 = decchar(c0, urlsafe);
|
||||
let v1: u8 = decchar(c1, urlsafe);
|
||||
if (v0 == 255u8) { return i: invalid; };
|
||||
if (v1 == 255u8) { return (i + 1): invalid; };
|
||||
// Last quad may carry '=' padding.
|
||||
if (i + 4 == end) {
|
||||
if (c2 == 61u8) {
|
||||
// "XX=="
|
||||
if (c3 != 61u8) { return (i + 3): invalid; };
|
||||
dst[j] = (v0 << 2u8) | (v1 >> 4u8);
|
||||
j += 1;
|
||||
i += 4;
|
||||
return j;
|
||||
};
|
||||
let v2: u8 = decchar(c2, urlsafe);
|
||||
if (v2 == 255u8) { return (i + 2): invalid; };
|
||||
if (c3 == 61u8) {
|
||||
// "XXX="
|
||||
dst[j] = (v0 << 2u8) | (v1 >> 4u8);
|
||||
dst[j + 1] = (v1 << 4u8) | (v2 >> 2u8);
|
||||
j += 2;
|
||||
i += 4;
|
||||
return j;
|
||||
};
|
||||
let v3: u8 = decchar(c3, urlsafe);
|
||||
if (v3 == 255u8) { return (i + 3): invalid; };
|
||||
dst[j] = (v0 << 2u8) | (v1 >> 4u8);
|
||||
dst[j + 1] = (v1 << 4u8) | (v2 >> 2u8);
|
||||
dst[j + 2] = (v2 << 6u8) | v3;
|
||||
j += 3;
|
||||
i += 4;
|
||||
return j;
|
||||
};
|
||||
let v2: u8 = decchar(c2, urlsafe);
|
||||
let v3: u8 = decchar(c3, urlsafe);
|
||||
if (v2 == 255u8) { return (i + 2): invalid; };
|
||||
if (v3 == 255u8) { return (i + 3): invalid; };
|
||||
dst[j] = (v0 << 2u8) | (v1 >> 4u8);
|
||||
dst[j + 1] = (v1 << 4u8) | (v2 >> 2u8);
|
||||
dst[j + 2] = (v2 << 6u8) | v3;
|
||||
i += 4;
|
||||
j += 3;
|
||||
};
|
||||
return j;
|
||||
};
|
||||
|
||||
// decode — decode std-alphabet base64 from `src` into `dst`. Returns
|
||||
// the count of decoded bytes, or invalid on a malformed input.
|
||||
export fn decode(dst: []u8, src: []u8) (i32 | invalid) = {
|
||||
return decodeinto(dst, src, false);
|
||||
};
|
||||
|
||||
// decodeurl — same as decode but accepts the url-safe alphabet.
|
||||
export fn decodeurl(dst: []u8, src: []u8) (i32 | invalid) = {
|
||||
return decodeinto(dst, src, true);
|
||||
};
|
||||
172
lib/encoding/base64/base64_test.ww
Normal file
172
lib/encoding/base64/base64_test.ww
Normal file
@@ -0,0 +1,172 @@
|
||||
use base64;
|
||||
|
||||
fn putstr(s: str, into: []u8, off: i32) i32 = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
into[off + i] = s[i];
|
||||
i += 1;
|
||||
};
|
||||
return off + s.len;
|
||||
};
|
||||
|
||||
fn streq(buf: []u8, expect: str) bool = {
|
||||
if (buf.len != expect.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < buf.len) {
|
||||
if (buf[i] != expect[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
fn encodevec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let m: i32 = base64.encode(outbuf[0:128], inbuf[0:n]);
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
};
|
||||
|
||||
@test fn rfc4648_vectors() void = {
|
||||
encodevec("", "");
|
||||
encodevec("f", "Zg==");
|
||||
encodevec("fo", "Zm8=");
|
||||
encodevec("foo", "Zm9v");
|
||||
encodevec("foob", "Zm9vYg==");
|
||||
encodevec("fooba", "Zm9vYmE=");
|
||||
encodevec("foobar", "Zm9vYmFy");
|
||||
};
|
||||
|
||||
fn decodevec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let r: (i32 | base64.invalid) = base64.decode(outbuf[0:128], inbuf[0:n]);
|
||||
match (r) {
|
||||
case let m: i32 => {
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
};
|
||||
|
||||
@test fn rfc4648_decode() void = {
|
||||
decodevec("", "");
|
||||
decodevec("Zg==", "f");
|
||||
decodevec("Zm8=", "fo");
|
||||
decodevec("Zm9v", "foo");
|
||||
decodevec("Zm9vYg==", "foob");
|
||||
decodevec("Zm9vYmE=", "fooba");
|
||||
decodevec("Zm9vYmFy", "foobar");
|
||||
};
|
||||
|
||||
@test fn alphabet_full() void = {
|
||||
// Round-trip every 6-bit value (0..63) by encoding three bytes that
|
||||
// expose b0=0x00, b1=AA, b2=FF — the encoded chars depend on all
|
||||
// four positions including the >>2 path.
|
||||
let i: i32 = 0;
|
||||
for (i < 64) {
|
||||
let bits: u8 = i: u8;
|
||||
// Construct a triple [bits<<2, 0, 0] so the first encoded
|
||||
// char encodes `bits`. The other three chars are derivable
|
||||
// from the remaining bytes; we only check the first here.
|
||||
let inbuf: [3]u8;
|
||||
inbuf[0] = bits << 2u8;
|
||||
inbuf[1] = 0u8;
|
||||
inbuf[2] = 0u8;
|
||||
let outbuf: [4]u8;
|
||||
let m: i32 = base64.encode(outbuf[0:4], inbuf[0:3]);
|
||||
if (m != 4) { let _: i32 = 1/0; };
|
||||
// Decoding back must give us `bits` in the high 6 bits of [0].
|
||||
let r: (i32 | base64.invalid) = base64.decode(inbuf[0:3], outbuf[0:4]);
|
||||
match (r) {
|
||||
case let n: i32 => {
|
||||
if (n != 3) { let _: i32 = 1/0; };
|
||||
if ((inbuf[0] >> 2u8) != bits) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
i += 1;
|
||||
};
|
||||
};
|
||||
|
||||
@test fn invalid_inputs() void = {
|
||||
let inbuf: [16]u8;
|
||||
let outbuf: [16]u8;
|
||||
// Length not a multiple of 4.
|
||||
let n: i32 = putstr("abc", inbuf[0:16], 0);
|
||||
let r1: (i32 | base64.invalid) = base64.decode(outbuf[0:16], inbuf[0:n]);
|
||||
match (r1) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base64.invalid => void;
|
||||
};
|
||||
// Bad char ('@' is not in the std alphabet).
|
||||
let n2: i32 = putstr("Z@==", inbuf[0:16], 0);
|
||||
let r2: (i32 | base64.invalid) = base64.decode(outbuf[0:16], inbuf[0:n2]);
|
||||
match (r2) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base64.invalid => void;
|
||||
};
|
||||
};
|
||||
|
||||
@test fn urlsafe_roundtrip() void = {
|
||||
// Byte sequence chosen so the std alphabet would use '+' and '/',
|
||||
// while url-safe replaces them with '-' and '_'. 0xFB = 11111011
|
||||
// hits index 62 in some quad, and 0xFF hits 63.
|
||||
let raw: [3]u8;
|
||||
raw[0] = 0xFBu8;
|
||||
raw[1] = 0xFFu8;
|
||||
raw[2] = 0xBFu8;
|
||||
let std: [8]u8;
|
||||
let url: [8]u8;
|
||||
let dec: [3]u8;
|
||||
let m1: i32 = base64.encode(std[0:8], raw[0:3]);
|
||||
let m2: i32 = base64.encodeurl(url[0:8], raw[0:3]);
|
||||
if (m1 != 4) { let _: i32 = 1/0; };
|
||||
if (m2 != 4) { let _: i32 = 1/0; };
|
||||
// Round-trip both ways.
|
||||
let r1: (i32 | base64.invalid) = base64.decode(dec[0:3], std[0:m1]);
|
||||
match (r1) {
|
||||
case let n: i32 => {
|
||||
if (n != 3) { let _: i32 = 1/0; };
|
||||
if (dec[0] != raw[0]) { let _: i32 = 1/0; };
|
||||
if (dec[1] != raw[1]) { let _: i32 = 1/0; };
|
||||
if (dec[2] != raw[2]) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
let r2: (i32 | base64.invalid) = base64.decodeurl(dec[0:3], url[0:m2]);
|
||||
match (r2) {
|
||||
case let n: i32 => {
|
||||
if (n != 3) { let _: i32 = 1/0; };
|
||||
if (dec[0] != raw[0]) { let _: i32 = 1/0; };
|
||||
if (dec[1] != raw[1]) { let _: i32 = 1/0; };
|
||||
if (dec[2] != raw[2]) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
};
|
||||
|
||||
@test fn sizes() void = {
|
||||
if (base64.encodedsize(0) != 0) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(1) != 4) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(2) != 4) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(3) != 4) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(4) != 8) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(6) != 8) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(7) != 12) { let _: i32 = 1/0; };
|
||||
if (base64.decodedsize(4) != 3) { let _: i32 = 1/0; };
|
||||
if (base64.decodedsize(8) != 6) { let _: i32 = 1/0; };
|
||||
};
|
||||
|
||||
export fn main() i32 = {
|
||||
rfc4648_vectors();
|
||||
rfc4648_decode();
|
||||
alphabet_full();
|
||||
invalid_inputs();
|
||||
urlsafe_roundtrip();
|
||||
sizes();
|
||||
return 0;
|
||||
};
|
||||
Reference in New Issue
Block a user