Shared miscompile in both stages — not a divergence. Bootstrap byte-id
passed throughout because both stages emitted the same wrong asm. Both
the C cgen (cmd/w6c/cgen.c TK_SLASH/TK_PERCENT) and the ww cgen
(selfhost/cmd/wcc/cgenexpr.ww) prepped IDIVQ with `MOVQ $0, DX`, which
is the unsigned 128-bit dividend shape. For a negative RAX, the CPU
then divides 2^64 + (-RAX) by the divisor — unsigned wraparound, not
signed division. Surfaced via lib/time/add() needing the verbatim Hare
signed-%-normalisation in ref/hare/time/arithm.ha.
Fix: emit CQO (sign-extend RAX into RDX:RAX, REX.W 99) on the signed
arm; keep MOVQ $0, DX on the unsigned arm where the DIVQ-vs-IDIVQ
dispatch was already correct. Since both stages always emit 64-bit
IDIVQ regardless of source width, a single CQO suffices for
i64/i32/i16/i8 — the dividend already lives in RAX sign-extended. No
CDQ/CWTL/CBTW needed.
Symmetric stages (rule 10): both stages were broken identically; both
get the same surgical fix. Adds A_CQO to each assembler's opcode set:
cstage in cmd/w6c/6.out.h + cmd/w6c/txt.c + cmd/w6a/{parse,asm}.c;
wwstage in selfhost/cmd/w6a/{types,parse,asm}.ww.
Class B (shared miscompile) — new in the session's polarity catalog.
Bootstrap byte-id is useless for catching it; semantic 9xx runtime
tests are the right shape. test/wcc/978_intdiv_signed.c covers 27 rows
× 2 drivers = 54 fixtures across {i8,i16,i32,i64,u8,u16,u32,u64} ×
{/, %} with width-boundary minima (INT8_MIN, INT16_MIN, INT32_MIN,
INT64_MIN/2) and high-bit-set unsigned anchors. INT64_MIN is spelled
(-INT64_MAX) - 1 per task #17 (wwstage NEGQ-over-imm drops digits on
-9223372036854775808i64); that literal-cgen bug is unrelated to this
fix.
Two known compound-assign workarounds at cmd/w6c/cgen.c:3765
(TK_SLASHEQ IDENT-local) and :3549 (TK_SLASHEQ/TK_PERCENTEQ
deref-compound) remain in tree; both depend on the assembler having
CQO, so they revert in a follow-up commit citing this one.
119 lines
2.6 KiB
C
119 lines
2.6 KiB
C
/*
|
|
* 6.out.h — amd64 instruction enum + register names. Mirrors the
|
|
* Plan 9 6c shape (cmd/6c/6.out.h) but trimmed to the subset that
|
|
* w6c emits and w6a consumes in this bootstrap. Each new opcode added
|
|
* here must also gain encoding support in cmd/w6a/asm.c.
|
|
*/
|
|
#ifndef SIX_OUT_H
|
|
#define SIX_OUT_H
|
|
|
|
/* registers — Plan 9 names; lowercase = 8-bit, etc. We use 64-bit. */
|
|
enum {
|
|
D_NONE = 0,
|
|
|
|
/* general purpose 64-bit */
|
|
D_AX, D_CX, D_DX, D_BX,
|
|
D_SP, D_BP, D_SI, D_DI,
|
|
D_R8, D_R9, D_R10, D_R11,
|
|
D_R12, D_R13, D_R14, D_R15,
|
|
|
|
/* SSE/XMM 64-bit float regs */
|
|
D_X0, D_X1, D_X2, D_X3,
|
|
D_X4, D_X5, D_X6, D_X7,
|
|
D_X8, D_X9, D_X10, D_X11,
|
|
D_X12, D_X13, D_X14, D_X15,
|
|
|
|
/* pseudo regs (Plan 9) */
|
|
D_PSP, /* SP pseudo (frame-relative) */
|
|
D_PFP, /* FP pseudo (incoming args) */
|
|
D_PSB, /* SB pseudo (static base) */
|
|
|
|
/* operand kinds; not registers but share the slot */
|
|
D_CONST, /* $N immediate */
|
|
D_BRANCH, /* label reference */
|
|
D_EXTERN, /* external symbol */
|
|
D_INDIR /* offset(reg) memory */
|
|
};
|
|
|
|
/* opcodes — the small set we currently emit & encode */
|
|
enum {
|
|
A_NOP = 0,
|
|
A_TEXT,
|
|
A_DATA,
|
|
A_DATAW, /* writable DATA: lands in .data (RW) instead of .text */
|
|
A_DATAR, /* reloc-only: patch a 64-bit slot in .data with a
|
|
* symbol's runtime VA. Pairs with a prior DATAW
|
|
* that left zero placeholder bytes. */
|
|
A_GLOBL,
|
|
A_END,
|
|
|
|
A_MOVQ,
|
|
A_MOVL,
|
|
A_MOVW,
|
|
A_MOVB,
|
|
A_MOVZBQ, /* movzx r64, r/m8 — byte zero-extended */
|
|
A_MOVZWQ, /* movzx r64, r/m16 — word zero-extended */
|
|
A_MOVSXD, /* movsxd r64, r/m32 — i32 sign-extended */
|
|
A_MOVSWQ, /* movsx r64, r/m16 — i16 sign-extended */
|
|
A_MOVSBQ, /* movsx r64, r/m8 — i8 sign-extended */
|
|
|
|
/* SSE2 scalar double-precision float */
|
|
A_MOVSD, /* xmm/m → xmm and xmm → m */
|
|
A_ADDSD,
|
|
A_SUBSD,
|
|
A_MULSD,
|
|
A_DIVSD,
|
|
A_UCOMISD,
|
|
A_CVTTSD2SI, /* truncate f64 → i64 */
|
|
A_CVTSI2SD, /* convert i64 → f64 */
|
|
|
|
/* SSE scalar single-precision float (f32). Same xmm regs. */
|
|
A_MOVSS,
|
|
A_ADDSS,
|
|
A_SUBSS,
|
|
A_MULSS,
|
|
A_DIVSS,
|
|
A_UCOMISS,
|
|
A_CVTTSS2SI,
|
|
A_CVTSI2SS,
|
|
A_CVTSD2SS, /* f64 → f32 truncate */
|
|
A_CVTSS2SD, /* f32 → f64 widen */
|
|
A_ADDQ,
|
|
A_SUBQ,
|
|
A_IMULQ,
|
|
A_IDIVQ,
|
|
A_DIVQ, /* unsigned 64-bit divide; sibling of IDIVQ */
|
|
A_CQO, /* sign-extend RAX into RDX:RAX; the signed-division
|
|
* prep that pairs with IDIVQ (DIVQ pairs with a
|
|
* MOVQ $0, DX zero-fill). */
|
|
A_NEGQ,
|
|
A_NOTQ,
|
|
A_ANDQ,
|
|
A_ORQ,
|
|
A_XORQ,
|
|
A_SHLQ,
|
|
A_SHRQ,
|
|
A_CMPQ,
|
|
|
|
A_PUSHQ,
|
|
A_POPQ,
|
|
A_LEAQ,
|
|
|
|
A_CALL,
|
|
A_RET,
|
|
A_JMP,
|
|
A_JE, A_JNE,
|
|
A_JL, A_JLE, A_JG, A_JGE,
|
|
A_JB, A_JBE, A_JA, A_JAE,
|
|
A_JZ, A_JNZ,
|
|
|
|
A_SYSCALL,
|
|
|
|
A_LAST
|
|
};
|
|
|
|
const char *anames(int); /* opcode -> mnemonic */
|
|
const char *rnames(int); /* register -> name */
|
|
|
|
#endif
|