w6c+wcc+selfhost+lib: int-cast truncate + use_alias, 5 new modules
Two cgen/check bugs surfaced by new lib modules, plus the modules
themselves (crc64, siphash, random, base64, base32).
1. `(big_u64): u32` (and `: u16`, `: u8`, `: bool`) didn't truncate.
N_CAST emitted nothing for int↔int; the value stayed in AX with
its upper bits intact and downstream CMPQ/DIVQ misread the slot.
The TK_TILDE path already had clamp logic for the same reason —
N_CAST was the missing case. Both stages now MOVL r,r for u32 and
ANDQ $mask for u8/u16/bool. Signed-narrow (i8/i16/i32) stays
no-op until w6a grows reg-reg MOVSBQ/MOVSWQ/MOVSXD. selfhost
cgcast walks alias chains via aliaslookup before checking
primsize/typenameisunsigned so `(u: random)` where
`type random = u64` still bypasses the clamp.
See cmd/w6c/cgen.c N_CAST and selfhost/cmd/wcc/cgenexpr.ww cgcast.
2. `mod.mod` type refs (`random.random` when the imported module
declares `export type random = u64;`) failed with "unknown type".
The driver concatenates imports into one flat scope, so SK_USE
`random` collided with SK_TYPE `random` and scope_define silently
dropped the use. resolve_typename's leaf lookup required
`kind == SK_USE` and gave up. Adds a `use_alias` flag to Sym; the
pass-1 decl scan now marks colliding syms in both directions
(use-after-type and type-after-use). resolve_typename and the
N_DOT cexpr branch treat `use_alias` like SK_USE for qualified
lookup. selfhost check.ww was already lenient on this path so no
ww-side change was needed; bootstrap fixed point (990-995) holds.
See cmd/wcc/check.c installdecl pass + N_DOT/resolve_typename and
cmd/wcc/ww.h Sym.use_alias.
New modules under lib/, each with @test vectors in *_test.ww and wired
into test/wcc/900_stdlib.c (26 modules → all compile):
- lib/hash/crc64 ECMA, ISO (mirror of crc32 shape)
- lib/hash/siphash SipHash-2-4, buffer-based sum/sum24
- lib/math/random SplitMix64 (init, next, u32n, u64n)
- lib/encoding/base64 RFC 4648 std + url-safe encode/decode + sizes
- lib/encoding/base32 RFC 4648 std + base32hex encode/decode + sizes
This commit is contained in:
186
lib/encoding/base32/base32.ww
Normal file
186
lib/encoding/base32/base32.ww
Normal file
@@ -0,0 +1,186 @@
|
||||
// encoding/base32 — RFC 4648 base32 encode/decode, buffer-based.
|
||||
//
|
||||
// Mirrors Hare's encoding::base32 surface, modulo Hare's stream-based
|
||||
// encoder/decoder. ww ships the in-memory subset only: `encode(dst,
|
||||
// src)` writes the encoded bytes into `dst`, returning the count;
|
||||
// `decode(dst, src)` writes the decoded bytes into `dst`, returning a
|
||||
// count or invalid.
|
||||
//
|
||||
// std uses 'A'-'Z' and '2'-'7' for indexes 0..31 (RFC 4648 §6); hex
|
||||
// uses '0'-'9' and 'A'-'V' (the base32hex alphabet, RFC 4648 §7).
|
||||
// Both pad encoded output with '=' to a multiple of 8 bytes.
|
||||
|
||||
export type invalid = !i32;
|
||||
|
||||
// encodedsize — bytes required to encode `n` source bytes (including
|
||||
// '=' padding). Hare names it the same.
|
||||
export fn encodedsize(n: i32) i32 = {
|
||||
if (n == 0) { return 0; };
|
||||
return ((n - 1) / 5 + 1) * 8;
|
||||
};
|
||||
|
||||
// decodedsize — upper bound on the number of bytes decoded from `n`
|
||||
// encoded bytes.
|
||||
export fn decodedsize(n: i32) i32 = {
|
||||
return (n / 8) * 5;
|
||||
};
|
||||
|
||||
// encchar — map a 5-bit value to its alphabet character. `hex` picks
|
||||
// the base32hex alphabet instead of std.
|
||||
fn encchar(v: u8, hex: bool) u8 = {
|
||||
if (hex) {
|
||||
if (v < 10u8) { return v + 48u8; }; // '0' + v
|
||||
return v + 55u8; // 'A' + (v - 10) = v + 55
|
||||
};
|
||||
if (v < 26u8) { return v + 65u8; }; // 'A' + v
|
||||
return v + 24u8; // '2' + (v - 26) = v + 24
|
||||
};
|
||||
|
||||
// decchar — inverse of encchar. Returns 0..31 or 255 on invalid char.
|
||||
// '=' is handled in the decode loop, not here.
|
||||
fn decchar(c: u8, hex: bool) u8 = {
|
||||
if (hex) {
|
||||
if (c >= 48u8) { if (c <= 57u8) { return c - 48u8; }; }; // '0'..'9'
|
||||
if (c >= 65u8) { if (c <= 86u8) { return c - 55u8; }; }; // 'A'..'V'
|
||||
return 255u8;
|
||||
};
|
||||
if (c >= 65u8) { if (c <= 90u8) { return c - 65u8; }; }; // 'A'..'Z'
|
||||
if (c >= 50u8) { if (c <= 55u8) { return c - 24u8; }; }; // '2'..'7'
|
||||
return 255u8;
|
||||
};
|
||||
|
||||
fn encgroup(dst: []u8, j: i32, src: []u8, i: i32, n: i32, hex: bool) void = {
|
||||
let b0: u8 = 0u8;
|
||||
let b1: u8 = 0u8;
|
||||
let b2: u8 = 0u8;
|
||||
let b3: u8 = 0u8;
|
||||
let b4: u8 = 0u8;
|
||||
if (n > 0) { b0 = src[i]; };
|
||||
if (n > 1) { b1 = src[i + 1]; };
|
||||
if (n > 2) { b2 = src[i + 2]; };
|
||||
if (n > 3) { b3 = src[i + 3]; };
|
||||
if (n > 4) { b4 = src[i + 4]; };
|
||||
dst[j] = encchar(b0 >> 3u8, hex);
|
||||
dst[j + 1] = encchar(((b0 & 7u8) << 2u8) | (b1 >> 6u8), hex);
|
||||
dst[j + 2] = encchar((b1 >> 1u8) & 31u8, hex);
|
||||
dst[j + 3] = encchar(((b1 & 1u8) << 4u8) | (b2 >> 4u8), hex);
|
||||
dst[j + 4] = encchar(((b2 & 15u8) << 1u8) | (b3 >> 7u8), hex);
|
||||
dst[j + 5] = encchar((b3 >> 2u8) & 31u8, hex);
|
||||
dst[j + 6] = encchar(((b3 & 3u8) << 3u8) | (b4 >> 5u8), hex);
|
||||
dst[j + 7] = encchar(b4 & 31u8, hex);
|
||||
// Pad the encoded slots that map past the source bytes.
|
||||
if (n < 5) {
|
||||
// n=1 → 2 chars then 6 '='. n=2 → 4 chars. n=3 → 5. n=4 → 7.
|
||||
let keep: i32 = 2;
|
||||
if (n == 2) { keep = 4; };
|
||||
if (n == 3) { keep = 5; };
|
||||
if (n == 4) { keep = 7; };
|
||||
let k: i32 = keep;
|
||||
for (k < 8) { dst[j + k] = 61u8; k += 1; }; // '='
|
||||
};
|
||||
};
|
||||
|
||||
fn encodeinto(dst: []u8, src: []u8, hex: bool) i32 = {
|
||||
let i: i32 = 0;
|
||||
let j: i32 = 0;
|
||||
for (i + 5 <= src.len) {
|
||||
encgroup(dst, j, src, i, 5, hex);
|
||||
i += 5;
|
||||
j += 8;
|
||||
};
|
||||
let rem: i32 = src.len - i;
|
||||
if (rem > 0) {
|
||||
encgroup(dst, j, src, i, rem, hex);
|
||||
j += 8;
|
||||
};
|
||||
return j;
|
||||
};
|
||||
|
||||
// encode — encode `src` into `dst` using the std (RFC 4648 §6)
|
||||
// alphabet. Returns bytes written. `dst` must hold at least
|
||||
// encodedsize(src.len) bytes.
|
||||
export fn encode(dst: []u8, src: []u8) i32 = {
|
||||
return encodeinto(dst, src, false);
|
||||
};
|
||||
|
||||
// encodehex — same as encode but uses the base32hex alphabet
|
||||
// (RFC 4648 §7).
|
||||
export fn encodehex(dst: []u8, src: []u8) i32 = {
|
||||
return encodeinto(dst, src, true);
|
||||
};
|
||||
|
||||
// padcount — number of bytes encoded in the last group, given the
|
||||
// count `p` of trailing '=' chars. RFC 4648 lists the legal mapping:
|
||||
// 6=>1, 4=>2, 3=>3, 1=>4, 0=>5. Returns -1 if `p` isn't legal.
|
||||
fn padcount(p: i32) i32 = {
|
||||
if (p == 0) { return 5; };
|
||||
if (p == 1) { return 4; };
|
||||
if (p == 3) { return 3; };
|
||||
if (p == 4) { return 2; };
|
||||
if (p == 6) { return 1; };
|
||||
return -1;
|
||||
};
|
||||
|
||||
fn decodeinto(dst: []u8, src: []u8, hex: bool) (i32 | invalid) = {
|
||||
if (src.len == 0) { return 0; };
|
||||
if ((src.len & 7) != 0) { return src.len: invalid; };
|
||||
let i: i32 = 0;
|
||||
let j: i32 = 0;
|
||||
let end: i32 = src.len;
|
||||
for (i < end) {
|
||||
let v: [8]u8;
|
||||
let last: bool = false;
|
||||
let p: i32 = 0;
|
||||
let k: i32 = 0;
|
||||
for (k < 8) {
|
||||
let c: u8 = src[i + k];
|
||||
if (c == 61u8) { // '='
|
||||
if (i + 8 != end) { return (i + k): invalid; };
|
||||
last = true;
|
||||
v[k] = 0u8;
|
||||
p += 1;
|
||||
} else {
|
||||
if (last) { return (i + k): invalid; };
|
||||
let d: u8 = decchar(c, hex);
|
||||
if (d == 255u8) { return (i + k): invalid; };
|
||||
v[k] = d;
|
||||
};
|
||||
k += 1;
|
||||
};
|
||||
let nb: i32 = 5;
|
||||
if (last) {
|
||||
nb = padcount(p);
|
||||
if (nb < 0) { return (i + 8 - p): invalid; };
|
||||
};
|
||||
// First two chars cover byte[0] (5 + 3 bits).
|
||||
if (nb > 0) {
|
||||
dst[j] = (v[0] << 3u8) | (v[1] >> 2u8);
|
||||
};
|
||||
if (nb > 1) {
|
||||
dst[j + 1] = (v[1] << 6u8) | (v[2] << 1u8) | (v[3] >> 4u8);
|
||||
};
|
||||
if (nb > 2) {
|
||||
dst[j + 2] = (v[3] << 4u8) | (v[4] >> 1u8);
|
||||
};
|
||||
if (nb > 3) {
|
||||
dst[j + 3] = (v[4] << 7u8) | (v[5] << 2u8) | (v[6] >> 3u8);
|
||||
};
|
||||
if (nb > 4) {
|
||||
dst[j + 4] = (v[6] << 5u8) | v[7];
|
||||
};
|
||||
j += nb;
|
||||
i += 8;
|
||||
};
|
||||
return j;
|
||||
};
|
||||
|
||||
// decode — decode std-alphabet base32 from `src` into `dst`. Returns
|
||||
// count of decoded bytes, or invalid on malformed input.
|
||||
export fn decode(dst: []u8, src: []u8) (i32 | invalid) = {
|
||||
return decodeinto(dst, src, false);
|
||||
};
|
||||
|
||||
// decodehex — same as decode but accepts base32hex.
|
||||
export fn decodehex(dst: []u8, src: []u8) (i32 | invalid) = {
|
||||
return decodeinto(dst, src, true);
|
||||
};
|
||||
155
lib/encoding/base32/base32_test.ww
Normal file
155
lib/encoding/base32/base32_test.ww
Normal file
@@ -0,0 +1,155 @@
|
||||
use base32;
|
||||
|
||||
fn putstr(s: str, into: []u8, off: i32) i32 = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
into[off + i] = s[i];
|
||||
i += 1;
|
||||
};
|
||||
return off + s.len;
|
||||
};
|
||||
|
||||
fn streq(buf: []u8, expect: str) bool = {
|
||||
if (buf.len != expect.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < buf.len) {
|
||||
if (buf[i] != expect[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
fn encvec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let m: i32 = base32.encode(outbuf[0:128], inbuf[0:n]);
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
};
|
||||
|
||||
@test fn rfc4648_std() void = {
|
||||
// RFC 4648 §10 test vectors.
|
||||
encvec("", "");
|
||||
encvec("f", "MY======");
|
||||
encvec("fo", "MZXQ====");
|
||||
encvec("foo", "MZXW6===");
|
||||
encvec("foob", "MZXW6YQ=");
|
||||
encvec("fooba", "MZXW6YTB");
|
||||
encvec("foobar", "MZXW6YTBOI======");
|
||||
};
|
||||
|
||||
fn decvec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let r: (i32 | base32.invalid) = base32.decode(outbuf[0:128], inbuf[0:n]);
|
||||
match (r) {
|
||||
case let m: i32 => {
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base32.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
};
|
||||
|
||||
@test fn rfc4648_decode() void = {
|
||||
decvec("", "");
|
||||
decvec("MY======", "f");
|
||||
decvec("MZXQ====", "fo");
|
||||
decvec("MZXW6===", "foo");
|
||||
decvec("MZXW6YQ=", "foob");
|
||||
decvec("MZXW6YTB", "fooba");
|
||||
decvec("MZXW6YTBOI======", "foobar");
|
||||
};
|
||||
|
||||
fn enchexvec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let m: i32 = base32.encodehex(outbuf[0:128], inbuf[0:n]);
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
};
|
||||
|
||||
@test fn rfc4648_hex() void = {
|
||||
// RFC 4648 §10 base32hex vectors.
|
||||
enchexvec("", "");
|
||||
enchexvec("f", "CO======");
|
||||
enchexvec("fo", "CPNG====");
|
||||
enchexvec("foo", "CPNMU===");
|
||||
enchexvec("foob", "CPNMUOG=");
|
||||
enchexvec("fooba", "CPNMUOJ1");
|
||||
enchexvec("foobar", "CPNMUOJ1E8======");
|
||||
};
|
||||
|
||||
@test fn roundtrip_all_quintets() void = {
|
||||
// Encode then decode every 5-byte combination of a small set.
|
||||
let raw: [5]u8;
|
||||
raw[0] = 0x00u8;
|
||||
raw[1] = 0x55u8;
|
||||
raw[2] = 0xAAu8;
|
||||
raw[3] = 0xFFu8;
|
||||
raw[4] = 0x01u8;
|
||||
let enc: [16]u8;
|
||||
let dec: [5]u8;
|
||||
let m: i32 = base32.encode(enc[0:16], raw[0:5]);
|
||||
if (m != 8) { let _: i32 = 1/0; };
|
||||
let r: (i32 | base32.invalid) = base32.decode(dec[0:5], enc[0:m]);
|
||||
match (r) {
|
||||
case let n: i32 => {
|
||||
if (n != 5) { let _: i32 = 1/0; };
|
||||
let i: i32 = 0;
|
||||
for (i < 5) {
|
||||
if (dec[i] != raw[i]) { let _: i32 = 1/0; };
|
||||
i += 1;
|
||||
};
|
||||
};
|
||||
case let e: base32.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
};
|
||||
|
||||
@test fn invalid_inputs() void = {
|
||||
let inbuf: [16]u8;
|
||||
let outbuf: [16]u8;
|
||||
// Length not a multiple of 8.
|
||||
let n: i32 = putstr("ABCD", inbuf[0:16], 0);
|
||||
let r1: (i32 | base32.invalid) = base32.decode(outbuf[0:16], inbuf[0:n]);
|
||||
match (r1) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base32.invalid => void;
|
||||
};
|
||||
// Bad pad count (5 '=' is illegal — must be 0,1,3,4,6).
|
||||
let n2: i32 = putstr("MZX=====", inbuf[0:16], 0);
|
||||
let r2: (i32 | base32.invalid) = base32.decode(outbuf[0:16], inbuf[0:n2]);
|
||||
match (r2) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base32.invalid => void;
|
||||
};
|
||||
// Bad char ('1' is not in the std alphabet).
|
||||
let n3: i32 = putstr("MZ1W6YTB", inbuf[0:16], 0);
|
||||
let r3: (i32 | base32.invalid) = base32.decode(outbuf[0:16], inbuf[0:n3]);
|
||||
match (r3) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base32.invalid => void;
|
||||
};
|
||||
};
|
||||
|
||||
@test fn sizes() void = {
|
||||
if (base32.encodedsize(0) != 0) { let _: i32 = 1/0; };
|
||||
if (base32.encodedsize(1) != 8) { let _: i32 = 1/0; };
|
||||
if (base32.encodedsize(5) != 8) { let _: i32 = 1/0; };
|
||||
if (base32.encodedsize(6) != 16) { let _: i32 = 1/0; };
|
||||
if (base32.decodedsize(8) != 5) { let _: i32 = 1/0; };
|
||||
if (base32.decodedsize(16) != 10) { let _: i32 = 1/0; };
|
||||
};
|
||||
|
||||
export fn main() i32 = {
|
||||
rfc4648_std();
|
||||
rfc4648_decode();
|
||||
rfc4648_hex();
|
||||
roundtrip_all_quintets();
|
||||
invalid_inputs();
|
||||
sizes();
|
||||
return 0;
|
||||
};
|
||||
Reference in New Issue
Block a user