w6c+w6a+selfhost+lib: cgen+asm bugs surfaced by hash modules
Seven fixes across the toolchain, plus three new lib/hash modules
(adler32, crc16, crc32) that surfaced them.
1. `~x` on u8/u16/u32 left the upper bits set: NOTQ inverts the
whole 64-bit register and nothing trimmed it back to type
width, so a returned `u16` would compare 64-bit against a
typed literal and disagree. Both stages now mask after NOTQ
for narrow unsigned: AND $0xFF/0xFFFF for u8/u16, MOVL r,r for
u32 (ANDQ $0xFFFFFFFF sign-extends imm32 and is a no-op).
Signed narrows stay sign-extended and need no fix-up. See
cmd/w6c/cgen.c N_UN TK_TILDE and selfhost cgenexpr.ww cgun
TK_TILDE with new nodeprimwidth helper.
2. w6a had no D_CONST immediate path for ANDQ / ORQ. cgen would
emit `ANDQ $65535, AX` and the rr encoder silently wrote
`21 /r` with garbage reg fields — the mask never happened.
Added `81 /4` (AND) and `81 /1` (OR) imm32 paths in both
cstage and selfhost w6a. The ~width fix above depends on this.
3. `s: []u8` cast as a direct fn argument produced a 0-length
slice. cgexpr for N_CAST left (AX=ptr, BX=len) from the str
source but never set CX (cap), and the arg-push fallback only
pushed AX. cgcast now synthesises CX=BX when target is slice
and source is str; node_isslice / arg-push recognise
cast-to-slice and emit the full (cap, len, ptr) triple. Both
stages.
4. `*[N]T` element-store used 8-byte stride + MOVQ regardless of
T's width. Indexing `buf: *[4]u16` would step 8 bytes and
write 8 bytes per element. Added idx_eff (drills *[N]T → T)
in cstage and the matching pointer-array drill in selfhost
elemsizeof. Also added MOVW / MOVZWQ / MOVSWQ to w6c, w6a,
and selfhost mirrors so 2-byte element stores/loads use the
right opcode (was falling through to MOVQ and trailing 6 bytes
into the next slot).
5. Slicing a top-level fixed array (`g[0:n]` where `g: [N]T` is
a global) computed the base from BP instead of the symbol —
localfind returned 0 and the cgen treated it as a local at
offset 0. Both N_SLICE-as-expression (cgslice) and N_SLICE-
as-call-arg paths now check let_islet / letvartnode and emit
LEAQ name(SB) when the base is a global array (or MOVQ
name(SB) for a global slice/pointer base). Both stages.
6. Top-level `let arr: [N]T = [v0, v1, ...]` link-failed on
cstage — emit_lets bailed when it saw N_ARRLIT init on an
array type, and the sz==8 scalar path then misemitted any
8-byte-sized array (e.g. [4]u16, [8]u8) as a single quad.
emit_lets now walks N_ARRLIT, evaluates each element as an
int/rune/bool/nil literal, packs per-element bytes
little-endian, and honours the trailing `...` repeat marker.
Selfhost already handled the literal-init path; fixed the
parallel sz==8 duplicate-DATAW emit on its side (the array
and the scalar paths both fired, last write winning at link
but the duplicate broke cross-stage byte-identicality on user
code with this shape).
7. w6a's per-line input buffer was a 1KB stack `char buf[1024]`.
A `DATAW` for a [256]u16 emits ~2080 bytes on one line, which
truncated mid-escape; the assembler then re-parsed the
remaining tail as garbage opcodes ("unknown opcode"). Bumped
cstage w6a to a 32K static buffer (selfhost w6a already
allocated per-line via amalloc).
lib: lib/hash/adler32, lib/hash/crc16, lib/hash/crc32 — pure
buffer-subset shape (matching lib/hash/fnv), with per-module
*_test.ww runnable via `ww test lib/hash/<name>`. Adler-32 plus
CRC-16 (CCITT/CMDA2000/DECT/ANSI) and CRC-32 (IEEE/Castagnoli/
Koopman) cover Hare's reference vectors bit-for-bit. Wired into
test/wcc/900_stdlib.c. .gitignore: lib/**/*.s,*.o so `ww test`
droppings stay untracked.
`make test` (26/26), `make bootstrap` (ww2≡ww3≡ww4), and per-module
`ww test` all pass. cgen output is byte-identical across cstage and
selfhost for every repro that previously diverged.
This commit is contained in:
@@ -453,6 +453,57 @@ a_encode(Asm *a)
|
||||
a->errs++;
|
||||
}
|
||||
break;
|
||||
case A_MOVW:
|
||||
/* 16-bit MOV: prefix 0x66 selects 16-bit operand size.
|
||||
* MOV r/m16, r16 — 66 89 /r; MOV r16, r/m16 — 66 8B /r.
|
||||
* No REX.W (operand-size prefix beats REX.W). */
|
||||
if (p->from.type >= D_AX && p->from.type <= D_R15
|
||||
&& p->to.type == D_INDIR) {
|
||||
a_emit_byte(a, 0x66);
|
||||
emit_rex(a, rhi(p->from.type), rhi(p->to.reg), 0);
|
||||
a_emit_byte(a, 0x89);
|
||||
emit_modrm_mem(a, rcode(p->from.type),
|
||||
p->to.reg, p->to.offset);
|
||||
} else if (p->from.type == D_INDIR
|
||||
&& p->to.type >= D_AX && p->to.type <= D_R15) {
|
||||
a_emit_byte(a, 0x66);
|
||||
emit_rex(a, rhi(p->to.type), rhi(p->from.reg), 0);
|
||||
a_emit_byte(a, 0x8B);
|
||||
emit_modrm_mem(a, rcode(p->to.type),
|
||||
p->from.reg, p->from.offset);
|
||||
} else {
|
||||
fprintf(stderr, "w6a: line %d: unsupported MOVW shape\n", p->line);
|
||||
a->errs++;
|
||||
}
|
||||
break;
|
||||
case A_MOVZWQ:
|
||||
/* MOVZX r64, r/m16 — 0F B7 /r with REX.W */
|
||||
if (p->from.type == D_INDIR
|
||||
&& p->to.type >= D_AX && p->to.type <= D_R15) {
|
||||
emit_rex(a, rhi(p->to.type), rhi(p->from.reg), 1);
|
||||
a_emit_byte(a, 0x0F);
|
||||
a_emit_byte(a, 0xB7);
|
||||
emit_modrm_mem(a, rcode(p->to.type),
|
||||
p->from.reg, p->from.offset);
|
||||
} else {
|
||||
fprintf(stderr, "w6a: line %d: unsupported MOVZWQ shape\n", p->line);
|
||||
a->errs++;
|
||||
}
|
||||
break;
|
||||
case A_MOVSWQ:
|
||||
/* MOVSX r64, r/m16 — 0F BF /r with REX.W */
|
||||
if (p->from.type == D_INDIR
|
||||
&& p->to.type >= D_AX && p->to.type <= D_R15) {
|
||||
emit_rex(a, rhi(p->to.type), rhi(p->from.reg), 1);
|
||||
a_emit_byte(a, 0x0F);
|
||||
a_emit_byte(a, 0xBF);
|
||||
emit_modrm_mem(a, rcode(p->to.type),
|
||||
p->from.reg, p->from.offset);
|
||||
} else {
|
||||
fprintf(stderr, "w6a: line %d: unsupported MOVSWQ shape\n", p->line);
|
||||
a->errs++;
|
||||
}
|
||||
break;
|
||||
case A_MOVB:
|
||||
/* MOV r/m8, r8 — 88 /r. No REX.W. We always emit REX
|
||||
* to allow access to SIL/DIL/BPL/SPL. */
|
||||
@@ -638,8 +689,24 @@ a_encode(Asm *a)
|
||||
else
|
||||
encode_rr(a, 0x29, p->from.type, p->to.type);
|
||||
break;
|
||||
case A_ANDQ: encode_rr(a, 0x21, p->from.type, p->to.type); break;
|
||||
case A_ORQ: encode_rr(a, 0x09, p->from.type, p->to.type); break;
|
||||
case A_ANDQ:
|
||||
/* AND r/m64, imm32 — 81 /4 (REX.W). Without the
|
||||
* D_CONST path the rr encoder would silently emit
|
||||
* a 0x21 with garbage reg fields. */
|
||||
if (p->from.type == D_CONST
|
||||
&& p->to.type >= D_AX && p->to.type <= D_R15)
|
||||
encode_ri_imm32(a, 0x81, 4, p->to.type, (i32)p->from.offset);
|
||||
else
|
||||
encode_rr(a, 0x21, p->from.type, p->to.type);
|
||||
break;
|
||||
case A_ORQ:
|
||||
/* OR r/m64, imm32 — 81 /1 (REX.W). Mirrors ANDQ. */
|
||||
if (p->from.type == D_CONST
|
||||
&& p->to.type >= D_AX && p->to.type <= D_R15)
|
||||
encode_ri_imm32(a, 0x81, 1, p->to.type, (i32)p->from.offset);
|
||||
else
|
||||
encode_rr(a, 0x09, p->from.type, p->to.type);
|
||||
break;
|
||||
case A_XORQ:
|
||||
if (p->from.type == D_CONST
|
||||
&& p->to.type >= D_AX && p->to.type <= D_R15)
|
||||
|
||||
@@ -83,8 +83,9 @@ opcode_lookup(const char *m)
|
||||
{
|
||||
struct { const char *m; int op; } tab[] = {
|
||||
{ "MOVQ", A_MOVQ }, { "MOVL", A_MOVL },
|
||||
{ "MOVB", A_MOVB }, { "MOVZBQ", A_MOVZBQ },
|
||||
{ "MOVSXD", A_MOVSXD },
|
||||
{ "MOVW", A_MOVW }, { "MOVB", A_MOVB },
|
||||
{ "MOVZBQ", A_MOVZBQ }, { "MOVZWQ", A_MOVZWQ },
|
||||
{ "MOVSXD", A_MOVSXD }, { "MOVSWQ", A_MOVSWQ },
|
||||
{ "MOVSD", A_MOVSD },
|
||||
{ "ADDSD", A_ADDSD },{ "SUBSD", A_SUBSD },
|
||||
{ "MULSD", A_MULSD },{ "DIVSD", A_DIVSD },
|
||||
@@ -263,7 +264,10 @@ parse_operand(Asm *a, const char *s, Aoperand *out)
|
||||
int
|
||||
a_parse(Asm *a)
|
||||
{
|
||||
char buf[1024];
|
||||
/* Big enough for a DATAW emitting a [256]u32 table (1024 bytes
|
||||
* → ~4100 chars of `\xNN` escapes plus directive boilerplate).
|
||||
* Selfhost w6a allocates per-line; this is the cstage equivalent. */
|
||||
static char buf[32768];
|
||||
char *line;
|
||||
size_t len;
|
||||
const char *pending_label = NULL;
|
||||
|
||||
Reference in New Issue
Block a user