lib/encoding/base64: Hare base64/base64url on io-streaming surface
Rewrite the buffer-based base64 placeholder as a faithful port of ref/hare/encoding/base64/base64.ha over the just-landed io-streaming surface (mirrors lib/encoding/hex). Ships: std_encoding/url_encoding (module-level `def` consts; decmap trailing 0xff run spelled out, no '...', to stay on #251 and avoid the #250 repeat-fill sugar); the streaming encoder newencoder/encode/ encodeslice/encodestr with a padding closer wired into the inline vtable; encodedsize/decodedsize; and decodestr as a direct in-memory decode via decmap (the same divergence hex took for its direct path — its return union carries errors.invalid, unconstrained by io.error). Deferred (at-site notes): the streaming decoder newdecoder/decode_reader (#247-sibling, blocked on #199b — io.error lacks errors.invalid). clear() wipes the work buffers with explicit full-length slices (`[0:len(...)]`) rather than Hare's bare-array decay (pending #258 [N]T->[]T coercion) to preserve the whole-array hygiene wipe. base64 graduates off 900_stdlib (cross-module refs resolve only via driver concatenation, as hex did); coverage at 984_base64_run over the RFC 4648 §10 vectors for std and url.
This commit is contained in:
@@ -1,174 +1,187 @@
|
||||
// base64_test — exercises lib/encoding/base64's io-streaming surface.
|
||||
// Run with `out/bin/ww run lib/encoding/base64/base64_test.ww`. Mirrors
|
||||
// Hare's base64 @test fns (ref/hare/encoding/base64/base64.ha:315,514,
|
||||
// 601) over the RFC 4648 §10 vectors, table-driven (parallel arrays;
|
||||
// tuple-row arrays are blocked by #111). Same signalled-then-fail()-
|
||||
// with-+10 pattern as the rest of the 9xx stdlib tests; non-zero exit
|
||||
// pinpoints the failing scenario.
|
||||
//
|
||||
// The streaming decoder (newdecoder) is deferred (#247-sibling), so the
|
||||
// decode side is exercised through decodestr only.
|
||||
|
||||
package base64;
|
||||
|
||||
import base64;
|
||||
import bytes;
|
||||
import errors;
|
||||
import io;
|
||||
import memio;
|
||||
import os;
|
||||
import strings;
|
||||
|
||||
fn putstr(s: str, into: []u8, off: i32) i32 = {
|
||||
let i: i32 = 0;
|
||||
for (i < s.len) {
|
||||
into[off + i] = s[i];
|
||||
i += 1;
|
||||
};
|
||||
return off + s.len;
|
||||
};
|
||||
let signalled: i32 = 0;
|
||||
fn fail() void = { os.exit(signalled + 10); };
|
||||
|
||||
fn streq(buf: []u8, expect: str) bool = {
|
||||
if (buf.len != expect.len) { return false; };
|
||||
fn streq(a: str, b: str) bool = {
|
||||
if (a.len != b.len) { return false; };
|
||||
let i: i32 = 0;
|
||||
for (i < buf.len) {
|
||||
if (buf[i] != expect[i]) { return false; };
|
||||
for (i < a.len) {
|
||||
if (a[i] != b[i]) { return false; };
|
||||
i += 1;
|
||||
};
|
||||
return true;
|
||||
};
|
||||
|
||||
fn encodevec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let m: i32 = base64.encode(outbuf[0:128], inbuf[0:n]);
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
};
|
||||
|
||||
@test fn rfc4648_vectors() void = {
|
||||
encodevec("", "");
|
||||
encodevec("f", "Zg==");
|
||||
encodevec("fo", "Zm8=");
|
||||
encodevec("foo", "Zm9v");
|
||||
encodevec("foob", "Zm9vYg==");
|
||||
encodevec("fooba", "Zm9vYmE=");
|
||||
encodevec("foobar", "Zm9vYmFy");
|
||||
};
|
||||
|
||||
fn decodevec(input: str, expect: str) void = {
|
||||
let inbuf: [128]u8;
|
||||
let outbuf: [128]u8;
|
||||
let n: i32 = putstr(input, inbuf[0:128], 0);
|
||||
let r: (i32 | base64.invalid) = base64.decode(outbuf[0:128], inbuf[0:n]);
|
||||
match (r) {
|
||||
case let m: i32 => {
|
||||
if (m != expect.len) { let _: i32 = 1/0; };
|
||||
if (!streq(outbuf[0:m], expect)) { let _: i32 = 1/0; };
|
||||
// enc_check — encode `raw` two ways (the io.handle sink via base64.encode
|
||||
// and the string form via base64.encodestr) and assert both equal
|
||||
// `expect`. ref/hare/encoding/base64/base64.ha:315.
|
||||
fn enc_check(enc: *base64.encoding, raw: []u8, expect: str) void = {
|
||||
let out: memio.stream = memio.dynamic();
|
||||
match (base64.encode(&out.vt, enc, raw)) {
|
||||
case let n: size => { if (n: i32 != raw.len) { fail(); }; };
|
||||
case let e: io.error => fail();
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
if (!streq(memio.string(&out), expect)) { fail(); };
|
||||
if (!streq(base64.encodestr(enc, raw), expect)) { fail(); };
|
||||
};
|
||||
|
||||
// dec_check — decodestr(`encoded`) must round-trip back to `raw`.
|
||||
// ref/hare/encoding/base64/base64.ha:514.
|
||||
fn dec_check(enc: *base64.encoding, encoded: str, raw: []u8) void = {
|
||||
match (base64.decodestr(enc, encoded)) {
|
||||
case let b: []u8 => { if (!bytes.equal(b, raw)) { fail(); }; };
|
||||
case let e: errors.invalid => fail();
|
||||
};
|
||||
};
|
||||
|
||||
@test fn rfc4648_decode() void = {
|
||||
decodevec("", "");
|
||||
decodevec("Zg==", "f");
|
||||
decodevec("Zm8=", "fo");
|
||||
decodevec("Zm9v", "foo");
|
||||
decodevec("Zm9vYg==", "foob");
|
||||
decodevec("Zm9vYmE=", "fooba");
|
||||
decodevec("Zm9vYmFy", "foobar");
|
||||
// inval_check — decodestr(`encoded`) must report errors.invalid.
|
||||
// ref/hare/encoding/base64/base64.ha:525.
|
||||
fn inval_check(enc: *base64.encoding, encoded: str) void = {
|
||||
match (base64.decodestr(enc, encoded)) {
|
||||
case let b: []u8 => fail();
|
||||
case let e: errors.invalid => void;
|
||||
};
|
||||
};
|
||||
|
||||
@test fn alphabet_full() void = {
|
||||
// Round-trip every 6-bit value (0..63) by encoding three bytes that
|
||||
// expose b0=0x00, b1=AA, b2=FF — the encoded chars depend on all
|
||||
// four positions including the >>2 path.
|
||||
// ---- RFC 4648 §10 vectors, encode + decodestr round-trip ----
|
||||
//
|
||||
// Inputs are the prefixes of "foobar". The §10 expected encodings
|
||||
// contain no '+' / '/', so std and base64url agree on these vectors —
|
||||
// both alphabets are driven over the same table here; the std-vs-url
|
||||
// distinctness chars are covered separately by urlsafe_distinct().
|
||||
|
||||
@test fn rfc4648_std() void = {
|
||||
let foobar: [6]u8 = ['f', 'o', 'o', 'b', 'a', 'r'];
|
||||
let exp: [7]str = ["", "Zg==", "Zm8=", "Zm9v", "Zm9vYg==",
|
||||
"Zm9vYmE=", "Zm9vYmFy"];
|
||||
let i: i32 = 0;
|
||||
for (i < 64) {
|
||||
let bits: u8 = i: u8;
|
||||
// Construct a triple [bits<<2, 0, 0] so the first encoded
|
||||
// char encodes `bits`. The other three chars are derivable
|
||||
// from the remaining bytes; we only check the first here.
|
||||
let inbuf: [3]u8;
|
||||
inbuf[0] = bits << 2u8;
|
||||
inbuf[1] = 0u8;
|
||||
inbuf[2] = 0u8;
|
||||
let outbuf: [4]u8;
|
||||
let m: i32 = base64.encode(outbuf[0:4], inbuf[0:3]);
|
||||
if (m != 4) { let _: i32 = 1/0; };
|
||||
// Decoding back must give us `bits` in the high 6 bits of [0].
|
||||
let r: (i32 | base64.invalid) = base64.decode(inbuf[0:3], outbuf[0:4]);
|
||||
match (r) {
|
||||
case let n: i32 => {
|
||||
if (n != 3) { let _: i32 = 1/0; };
|
||||
if ((inbuf[0] >> 2u8) != bits) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
for (i <= 6) {
|
||||
enc_check(&base64.std_encoding, foobar[0:i], exp[i]);
|
||||
dec_check(&base64.std_encoding, exp[i], foobar[0:i]);
|
||||
i += 1;
|
||||
};
|
||||
};
|
||||
|
||||
@test fn invalid_inputs() void = {
|
||||
let inbuf: [16]u8;
|
||||
let outbuf: [16]u8;
|
||||
// Length not a multiple of 4.
|
||||
let n: i32 = putstr("abc", inbuf[0:16], 0);
|
||||
let r1: (i32 | base64.invalid) = base64.decode(outbuf[0:16], inbuf[0:n]);
|
||||
match (r1) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base64.invalid => void;
|
||||
};
|
||||
// Bad char ('@' is not in the std alphabet).
|
||||
let n2: i32 = putstr("Z@==", inbuf[0:16], 0);
|
||||
let r2: (i32 | base64.invalid) = base64.decode(outbuf[0:16], inbuf[0:n2]);
|
||||
match (r2) {
|
||||
case let m: i32 => { let _: i32 = 1/0; };
|
||||
case let e: base64.invalid => void;
|
||||
@test fn rfc4648_url() void = {
|
||||
let foobar: [6]u8 = ['f', 'o', 'o', 'b', 'a', 'r'];
|
||||
let exp: [7]str = ["", "Zg==", "Zm8=", "Zm9v", "Zm9vYg==",
|
||||
"Zm9vYmE=", "Zm9vYmFy"];
|
||||
let i: i32 = 0;
|
||||
for (i <= 6) {
|
||||
enc_check(&base64.url_encoding, foobar[0:i], exp[i]);
|
||||
dec_check(&base64.url_encoding, exp[i], foobar[0:i]);
|
||||
i += 1;
|
||||
};
|
||||
};
|
||||
|
||||
@test fn urlsafe_roundtrip() void = {
|
||||
// Byte sequence chosen so the std alphabet would use '+' and '/',
|
||||
// while url-safe replaces them with '-' and '_'. 0xFB = 11111011
|
||||
// hits index 62 in some quad, and 0xFF hits 63.
|
||||
let raw: [3]u8;
|
||||
raw[0] = 0xFBu8;
|
||||
raw[1] = 0xFFu8;
|
||||
raw[2] = 0xBFu8;
|
||||
let std: [8]u8;
|
||||
let url: [8]u8;
|
||||
let dec: [3]u8;
|
||||
let m1: i32 = base64.encode(std[0:8], raw[0:3]);
|
||||
let m2: i32 = base64.encodeurl(url[0:8], raw[0:3]);
|
||||
if (m1 != 4) { let _: i32 = 1/0; };
|
||||
if (m2 != 4) { let _: i32 = 1/0; };
|
||||
// Round-trip both ways.
|
||||
let r1: (i32 | base64.invalid) = base64.decode(dec[0:3], std[0:m1]);
|
||||
match (r1) {
|
||||
case let n: i32 => {
|
||||
if (n != 3) { let _: i32 = 1/0; };
|
||||
if (dec[0] != raw[0]) { let _: i32 = 1/0; };
|
||||
if (dec[1] != raw[1]) { let _: i32 = 1/0; };
|
||||
if (dec[2] != raw[2]) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
let r2: (i32 | base64.invalid) = base64.decodeurl(dec[0:3], url[0:m2]);
|
||||
match (r2) {
|
||||
case let n: i32 => {
|
||||
if (n != 3) { let _: i32 = 1/0; };
|
||||
if (dec[0] != raw[0]) { let _: i32 = 1/0; };
|
||||
if (dec[1] != raw[1]) { let _: i32 = 1/0; };
|
||||
if (dec[2] != raw[2]) { let _: i32 = 1/0; };
|
||||
};
|
||||
case let e: base64.invalid => { let _: i32 = 1/0; };
|
||||
};
|
||||
// ---- std vs base64url alphabet distinctness ----
|
||||
//
|
||||
// [0xFB, 0xFF, 0xBF] hits the 62/63 alphabet slots: std emits '+'/'/',
|
||||
// url emits '-'/'_'. The two encodings must differ, each round-trips
|
||||
// under its own alphabet, and each is INVALID under the other (std
|
||||
// decmap marks '-'/'_' 0xff and url marks '+'/'/' 0xff).
|
||||
|
||||
@test fn urlsafe_distinct() void = {
|
||||
let raw: [3]u8 = [0xFBu8, 0xFFu8, 0xBFu8];
|
||||
let s_std: str = base64.encodestr(&base64.std_encoding, raw[0:3]);
|
||||
let s_url: str = base64.encodestr(&base64.url_encoding, raw[0:3]);
|
||||
if (streq(s_std, s_url)) { fail(); };
|
||||
|
||||
dec_check(&base64.std_encoding, s_std, raw[0:3]);
|
||||
dec_check(&base64.url_encoding, s_url, raw[0:3]);
|
||||
|
||||
// cross-alphabet decode must reject the foreign chars.
|
||||
inval_check(&base64.std_encoding, s_url);
|
||||
inval_check(&base64.url_encoding, s_std);
|
||||
};
|
||||
|
||||
// ---- decodestr error cases ---- ref/hare/encoding/base64/base64.ha:525
|
||||
//
|
||||
// Wrong length, bad char, embedded / excess padding all → invalid.
|
||||
|
||||
@test fn decode_invalid() void = {
|
||||
inval_check(&base64.std_encoding, "Zg"); // not a multiple of 4
|
||||
inval_check(&base64.std_encoding, "Z@=="); // '@' not in alphabet
|
||||
inval_check(&base64.std_encoding, "===="); // all padding
|
||||
inval_check(&base64.std_encoding, "Zg==Zg=="); // data after padding
|
||||
inval_check(&base64.std_encoding, "Zm8=Zm8="); // data after padding
|
||||
inval_check(&base64.std_encoding, "@Zg="); // bad leading char
|
||||
};
|
||||
|
||||
// ---- size calc ---- ref/hare/encoding/base64/base64.ha:601
|
||||
|
||||
@test fn sizes() void = {
|
||||
if (base64.encodedsize(0) != 0) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(1) != 4) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(2) != 4) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(3) != 4) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(4) != 8) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(6) != 8) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(7) != 12) { let _: i32 = 1/0; };
|
||||
if (base64.decodedsize(4) != 3) { let _: i32 = 1/0; };
|
||||
if (base64.decodedsize(8) != 6) { let _: i32 = 1/0; };
|
||||
if (base64.encodedsize(0) != 0) { fail(); };
|
||||
if (base64.encodedsize(1) != 4) { fail(); };
|
||||
if (base64.encodedsize(2) != 4) { fail(); };
|
||||
if (base64.encodedsize(3) != 4) { fail(); };
|
||||
if (base64.encodedsize(4) != 8) { fail(); };
|
||||
if (base64.encodedsize(10) != 16) { fail(); };
|
||||
if (base64.encodedsize(119) != 160) { fail(); };
|
||||
if (base64.encodedsize(120) != 160) { fail(); };
|
||||
if (base64.encodedsize(121) != 164) { fail(); };
|
||||
if (base64.encodedsize(122) != 164) { fail(); };
|
||||
if (base64.encodedsize(123) != 164) { fail(); };
|
||||
if (base64.decodedsize(0) != 0) { fail(); };
|
||||
if (base64.decodedsize(4) != 3) { fail(); };
|
||||
if (base64.decodedsize(8) != 6) { fail(); };
|
||||
if (base64.decodedsize(160) != 120) { fail(); };
|
||||
if (base64.decodedsize(164) != 123) { fail(); };
|
||||
};
|
||||
|
||||
// ---- round-trip every byte value 0..255 (std + url) ----
|
||||
|
||||
@test fn roundtrip_all_bytes() void = {
|
||||
let src: [256]u8;
|
||||
let i: i32 = 0;
|
||||
for (i < 256) {
|
||||
src[i] = i: u8;
|
||||
i += 1;
|
||||
};
|
||||
let s: str = base64.encodestr(&base64.std_encoding, src[0:256]);
|
||||
match (base64.decodestr(&base64.std_encoding, s)) {
|
||||
case let b: []u8 => {
|
||||
if (b.len != 256) { fail(); };
|
||||
if (!bytes.equal(b, src[0:256])) { fail(); };
|
||||
};
|
||||
case let e: errors.invalid => fail();
|
||||
};
|
||||
let u: str = base64.encodestr(&base64.url_encoding, src[0:256]);
|
||||
match (base64.decodestr(&base64.url_encoding, u)) {
|
||||
case let b: []u8 => {
|
||||
if (b.len != 256) { fail(); };
|
||||
if (!bytes.equal(b, src[0:256])) { fail(); };
|
||||
};
|
||||
case let e: errors.invalid => fail();
|
||||
};
|
||||
};
|
||||
|
||||
export fn main() i32 = {
|
||||
rfc4648_vectors();
|
||||
rfc4648_decode();
|
||||
alphabet_full();
|
||||
invalid_inputs();
|
||||
urlsafe_roundtrip();
|
||||
sizes();
|
||||
signalled = 1; rfc4648_std();
|
||||
signalled = 2; rfc4648_url();
|
||||
signalled = 3; urlsafe_distinct();
|
||||
signalled = 4; decode_invalid();
|
||||
signalled = 5; sizes();
|
||||
signalled = 6; roundtrip_all_bytes();
|
||||
return 0;
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user