lib/encoding/base64: Hare base64/base64url on io-streaming surface

Rewrite the buffer-based base64 placeholder as a faithful port of
ref/hare/encoding/base64/base64.ha over the just-landed io-streaming
surface (mirrors lib/encoding/hex).

Ships: std_encoding/url_encoding (module-level `def` consts; decmap
trailing 0xff run spelled out, no '...', to stay on #251 and avoid the
#250 repeat-fill sugar); the streaming encoder newencoder/encode/
encodeslice/encodestr with a padding closer wired into the inline
vtable; encodedsize/decodedsize; and decodestr as a direct in-memory
decode via decmap (the same divergence hex took for its direct path —
its return union carries errors.invalid, unconstrained by io.error).

Deferred (at-site notes): the streaming decoder newdecoder/decode_reader
(#247-sibling, blocked on #199b — io.error lacks errors.invalid).

clear() wipes the work buffers with explicit full-length slices
(`[0:len(...)]`) rather than Hare's bare-array decay (pending #258
[N]T->[]T coercion) to preserve the whole-array hygiene wipe.

base64 graduates off 900_stdlib (cross-module refs resolve only via
driver concatenation, as hex did); coverage at 984_base64_run over the
RFC 4648 §10 vectors for std and url.
This commit is contained in:
2026-06-02 04:55:08 +09:00
parent ca8c78e97d
commit 4d0d3b58e6
3 changed files with 566 additions and 312 deletions

View File

@@ -1,174 +1,187 @@
// base64_test — exercises lib/encoding/base64's io-streaming surface.
// Run with `out/bin/ww run lib/encoding/base64/base64_test.ww`. Mirrors
// Hare's base64 @test fns (ref/hare/encoding/base64/base64.ha:315,514,
// 601) over the RFC 4648 §10 vectors, table-driven (parallel arrays;
// tuple-row arrays are blocked by #111). Same signalled-then-fail()-
// with-+10 pattern as the rest of the 9xx stdlib tests; non-zero exit
// pinpoints the failing scenario.
//
// The streaming decoder (newdecoder) is deferred (#247-sibling), so the
// decode side is exercised through decodestr only.
package base64;
import base64;
import bytes;
import errors;
import io;
import memio;
import os;
import strings;
fn putstr(s: str, into: []u8, off: i32) i32 = {
let i: i32 = 0;
for (i < s.len) {
into[off + i] = s[i];
i += 1;
};
return off + s.len;
};
let signalled: i32 = 0;
fn fail() void = { os.exit(signalled + 10); };
fn streq(buf: []u8, expect: str) bool = {
if (buf.len != expect.len) { return false; };
fn streq(a: str, b: str) bool = {
if (a.len != b.len) { return false; };
let i: i32 = 0;
for (i < buf.len) {
if (buf[i] != expect[i]) { return false; };
for (i < a.len) {
if (a[i] != b[i]) { return false; };
i += 1;
};
return true;
};
fn encodevec(input: str, expect: str) void = {
let inbuf: [128]u8;
let outbuf: [128]u8;
let n: i32 = putstr(input, inbuf[0:128], 0);
let m: i32 = base64.encode(outbuf[0:128], inbuf[0:n]);
if (m != expect.len) { let _: i32 = 1/0; };
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
};
@test fn rfc4648_vectors() void = {
encodevec("", "");
encodevec("f", "Zg==");
encodevec("fo", "Zm8=");
encodevec("foo", "Zm9v");
encodevec("foob", "Zm9vYg==");
encodevec("fooba", "Zm9vYmE=");
encodevec("foobar", "Zm9vYmFy");
};
fn decodevec(input: str, expect: str) void = {
let inbuf: [128]u8;
let outbuf: [128]u8;
let n: i32 = putstr(input, inbuf[0:128], 0);
let r: (i32 | base64.invalid) = base64.decode(outbuf[0:128], inbuf[0:n]);
match (r) {
case let m: i32 => {
if (m != expect.len) { let _: i32 = 1/0; };
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
// enc_check — encode `raw` two ways (the io.handle sink via base64.encode
// and the string form via base64.encodestr) and assert both equal
// `expect`. ref/hare/encoding/base64/base64.ha:315.
fn enc_check(enc: *base64.encoding, raw: []u8, expect: str) void = {
let out: memio.stream = memio.dynamic();
match (base64.encode(&out.vt, enc, raw)) {
case let n: size => { if (n: i32 != raw.len) { fail(); }; };
case let e: io.error => fail();
};
case let e: base64.invalid => { let _: i32 = 1/0; };
if (!streq(memio.string(&out), expect)) { fail(); };
if (!streq(base64.encodestr(enc, raw), expect)) { fail(); };
};
// dec_check — decodestr(`encoded`) must round-trip back to `raw`.
// ref/hare/encoding/base64/base64.ha:514.
fn dec_check(enc: *base64.encoding, encoded: str, raw: []u8) void = {
match (base64.decodestr(enc, encoded)) {
case let b: []u8 => { if (!bytes.equal(b, raw)) { fail(); }; };
case let e: errors.invalid => fail();
};
};
@test fn rfc4648_decode() void = {
decodevec("", "");
decodevec("Zg==", "f");
decodevec("Zm8=", "fo");
decodevec("Zm9v", "foo");
decodevec("Zm9vYg==", "foob");
decodevec("Zm9vYmE=", "fooba");
decodevec("Zm9vYmFy", "foobar");
// inval_check — decodestr(`encoded`) must report errors.invalid.
// ref/hare/encoding/base64/base64.ha:525.
fn inval_check(enc: *base64.encoding, encoded: str) void = {
match (base64.decodestr(enc, encoded)) {
case let b: []u8 => fail();
case let e: errors.invalid => void;
};
};
@test fn alphabet_full() void = {
// Round-trip every 6-bit value (0..63) by encoding three bytes that
// expose b0=0x00, b1=AA, b2=FF — the encoded chars depend on all
// four positions including the >>2 path.
// ---- RFC 4648 §10 vectors, encode + decodestr round-trip ----
//
// Inputs are the prefixes of "foobar". The §10 expected encodings
// contain no '+' / '/', so std and base64url agree on these vectors —
// both alphabets are driven over the same table here; the std-vs-url
// distinctness chars are covered separately by urlsafe_distinct().
@test fn rfc4648_std() void = {
let foobar: [6]u8 = ['f', 'o', 'o', 'b', 'a', 'r'];
let exp: [7]str = ["", "Zg==", "Zm8=", "Zm9v", "Zm9vYg==",
"Zm9vYmE=", "Zm9vYmFy"];
let i: i32 = 0;
for (i < 64) {
let bits: u8 = i: u8;
// Construct a triple [bits<<2, 0, 0] so the first encoded
// char encodes `bits`. The other three chars are derivable
// from the remaining bytes; we only check the first here.
let inbuf: [3]u8;
inbuf[0] = bits << 2u8;
inbuf[1] = 0u8;
inbuf[2] = 0u8;
let outbuf: [4]u8;
let m: i32 = base64.encode(outbuf[0:4], inbuf[0:3]);
if (m != 4) { let _: i32 = 1/0; };
// Decoding back must give us `bits` in the high 6 bits of [0].
let r: (i32 | base64.invalid) = base64.decode(inbuf[0:3], outbuf[0:4]);
match (r) {
case let n: i32 => {
if (n != 3) { let _: i32 = 1/0; };
if ((inbuf[0] >> 2u8) != bits) { let _: i32 = 1/0; };
};
case let e: base64.invalid => { let _: i32 = 1/0; };
};
for (i <= 6) {
enc_check(&base64.std_encoding, foobar[0:i], exp[i]);
dec_check(&base64.std_encoding, exp[i], foobar[0:i]);
i += 1;
};
};
@test fn invalid_inputs() void = {
let inbuf: [16]u8;
let outbuf: [16]u8;
// Length not a multiple of 4.
let n: i32 = putstr("abc", inbuf[0:16], 0);
let r1: (i32 | base64.invalid) = base64.decode(outbuf[0:16], inbuf[0:n]);
match (r1) {
case let m: i32 => { let _: i32 = 1/0; };
case let e: base64.invalid => void;
};
// Bad char ('@' is not in the std alphabet).
let n2: i32 = putstr("Z@==", inbuf[0:16], 0);
let r2: (i32 | base64.invalid) = base64.decode(outbuf[0:16], inbuf[0:n2]);
match (r2) {
case let m: i32 => { let _: i32 = 1/0; };
case let e: base64.invalid => void;
@test fn rfc4648_url() void = {
let foobar: [6]u8 = ['f', 'o', 'o', 'b', 'a', 'r'];
let exp: [7]str = ["", "Zg==", "Zm8=", "Zm9v", "Zm9vYg==",
"Zm9vYmE=", "Zm9vYmFy"];
let i: i32 = 0;
for (i <= 6) {
enc_check(&base64.url_encoding, foobar[0:i], exp[i]);
dec_check(&base64.url_encoding, exp[i], foobar[0:i]);
i += 1;
};
};
@test fn urlsafe_roundtrip() void = {
// Byte sequence chosen so the std alphabet would use '+' and '/',
// while url-safe replaces them with '-' and '_'. 0xFB = 11111011
// hits index 62 in some quad, and 0xFF hits 63.
let raw: [3]u8;
raw[0] = 0xFBu8;
raw[1] = 0xFFu8;
raw[2] = 0xBFu8;
let std: [8]u8;
let url: [8]u8;
let dec: [3]u8;
let m1: i32 = base64.encode(std[0:8], raw[0:3]);
let m2: i32 = base64.encodeurl(url[0:8], raw[0:3]);
if (m1 != 4) { let _: i32 = 1/0; };
if (m2 != 4) { let _: i32 = 1/0; };
// Round-trip both ways.
let r1: (i32 | base64.invalid) = base64.decode(dec[0:3], std[0:m1]);
match (r1) {
case let n: i32 => {
if (n != 3) { let _: i32 = 1/0; };
if (dec[0] != raw[0]) { let _: i32 = 1/0; };
if (dec[1] != raw[1]) { let _: i32 = 1/0; };
if (dec[2] != raw[2]) { let _: i32 = 1/0; };
};
case let e: base64.invalid => { let _: i32 = 1/0; };
};
let r2: (i32 | base64.invalid) = base64.decodeurl(dec[0:3], url[0:m2]);
match (r2) {
case let n: i32 => {
if (n != 3) { let _: i32 = 1/0; };
if (dec[0] != raw[0]) { let _: i32 = 1/0; };
if (dec[1] != raw[1]) { let _: i32 = 1/0; };
if (dec[2] != raw[2]) { let _: i32 = 1/0; };
};
case let e: base64.invalid => { let _: i32 = 1/0; };
};
// ---- std vs base64url alphabet distinctness ----
//
// [0xFB, 0xFF, 0xBF] hits the 62/63 alphabet slots: std emits '+'/'/',
// url emits '-'/'_'. The two encodings must differ, each round-trips
// under its own alphabet, and each is INVALID under the other (std
// decmap marks '-'/'_' 0xff and url marks '+'/'/' 0xff).
@test fn urlsafe_distinct() void = {
let raw: [3]u8 = [0xFBu8, 0xFFu8, 0xBFu8];
let s_std: str = base64.encodestr(&base64.std_encoding, raw[0:3]);
let s_url: str = base64.encodestr(&base64.url_encoding, raw[0:3]);
if (streq(s_std, s_url)) { fail(); };
dec_check(&base64.std_encoding, s_std, raw[0:3]);
dec_check(&base64.url_encoding, s_url, raw[0:3]);
// cross-alphabet decode must reject the foreign chars.
inval_check(&base64.std_encoding, s_url);
inval_check(&base64.url_encoding, s_std);
};
// ---- decodestr error cases ---- ref/hare/encoding/base64/base64.ha:525
//
// Wrong length, bad char, embedded / excess padding all → invalid.
@test fn decode_invalid() void = {
inval_check(&base64.std_encoding, "Zg"); // not a multiple of 4
inval_check(&base64.std_encoding, "Z@=="); // '@' not in alphabet
inval_check(&base64.std_encoding, "===="); // all padding
inval_check(&base64.std_encoding, "Zg==Zg=="); // data after padding
inval_check(&base64.std_encoding, "Zm8=Zm8="); // data after padding
inval_check(&base64.std_encoding, "@Zg="); // bad leading char
};
// ---- size calc ---- ref/hare/encoding/base64/base64.ha:601
@test fn sizes() void = {
if (base64.encodedsize(0) != 0) { let _: i32 = 1/0; };
if (base64.encodedsize(1) != 4) { let _: i32 = 1/0; };
if (base64.encodedsize(2) != 4) { let _: i32 = 1/0; };
if (base64.encodedsize(3) != 4) { let _: i32 = 1/0; };
if (base64.encodedsize(4) != 8) { let _: i32 = 1/0; };
if (base64.encodedsize(6) != 8) { let _: i32 = 1/0; };
if (base64.encodedsize(7) != 12) { let _: i32 = 1/0; };
if (base64.decodedsize(4) != 3) { let _: i32 = 1/0; };
if (base64.decodedsize(8) != 6) { let _: i32 = 1/0; };
if (base64.encodedsize(0) != 0) { fail(); };
if (base64.encodedsize(1) != 4) { fail(); };
if (base64.encodedsize(2) != 4) { fail(); };
if (base64.encodedsize(3) != 4) { fail(); };
if (base64.encodedsize(4) != 8) { fail(); };
if (base64.encodedsize(10) != 16) { fail(); };
if (base64.encodedsize(119) != 160) { fail(); };
if (base64.encodedsize(120) != 160) { fail(); };
if (base64.encodedsize(121) != 164) { fail(); };
if (base64.encodedsize(122) != 164) { fail(); };
if (base64.encodedsize(123) != 164) { fail(); };
if (base64.decodedsize(0) != 0) { fail(); };
if (base64.decodedsize(4) != 3) { fail(); };
if (base64.decodedsize(8) != 6) { fail(); };
if (base64.decodedsize(160) != 120) { fail(); };
if (base64.decodedsize(164) != 123) { fail(); };
};
// ---- round-trip every byte value 0..255 (std + url) ----
@test fn roundtrip_all_bytes() void = {
let src: [256]u8;
let i: i32 = 0;
for (i < 256) {
src[i] = i: u8;
i += 1;
};
let s: str = base64.encodestr(&base64.std_encoding, src[0:256]);
match (base64.decodestr(&base64.std_encoding, s)) {
case let b: []u8 => {
if (b.len != 256) { fail(); };
if (!bytes.equal(b, src[0:256])) { fail(); };
};
case let e: errors.invalid => fail();
};
let u: str = base64.encodestr(&base64.url_encoding, src[0:256]);
match (base64.decodestr(&base64.url_encoding, u)) {
case let b: []u8 => {
if (b.len != 256) { fail(); };
if (!bytes.equal(b, src[0:256])) { fail(); };
};
case let e: errors.invalid => fail();
};
};
export fn main() i32 = {
rfc4648_vectors();
rfc4648_decode();
alphabet_full();
invalid_inputs();
urlsafe_roundtrip();
sizes();
signalled = 1; rfc4648_std();
signalled = 2; rfc4648_url();
signalled = 3; urlsafe_distinct();
signalled = 4; decode_invalid();
signalled = 5; sizes();
signalled = 6; roundtrip_all_bytes();
return 0;
};