Files
Odin/core/rexcode/isa/arm64/operands.odin
Flāvius fc1be3902d rexcode/arm64: merge system registers into Register
Register widens to a u32: hw number in bits 0-4, class byte in bits 8-15 --
bit-identical to the old u16 layout -- and, for the new REG_SYS class only,
the 15-bit MRS/MSR field in bits 16-30. System_Register, its Operand_Kind,
and the union's sysreg member are gone; a system register is now a plain
.REGISTER operand distinguished by class, so it flows through matching,
packing, and printing like any other register.

Memory is untouched: every class legal in an address still lives entirely in
the low 16 bits of the u32, so its 16-bit register slots stay lossless and
NONE round-trips (verified: Odin bit_fields zero-extend on read and reject
overflowing constants at compile time). Operand stays 11 bytes -- the union's
largest member is still 8 -- and Instruction stays exactly 64.

op_sysreg survives as an op_reg alias so MRS/MSR call sites read as what
they are, and the sysreg constants keep their field value in their name:
NZCV is now Register(0x5A10_1000) where it was System_Register(0x5A10).

New pipeline test: every SYSREG_NAMES entry round-trips encode -> decode ->
print byte-exactly, with the expected word derived from the table value.
All 330 table + 134 pipeline checks pass; benchmarks show encode ~3% faster,
decode and print at parity.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01AFeLCDKi5kRMtHrskUaRfw
2026-08-30 18:23:06 -04:00

462 lines
17 KiB
Odin

// rexcode · Brendan Punsky (dotbmp@github), original author
package rexcode_arm64
// =============================================================================
// AArch64 OPERANDS
// =============================================================================
//
// AArch64 has a rich addressing repertoire:
//
// [Xn] OFFSET with imm=0
// [Xn, #imm] OFFSET (signed 9 or unsigned scaled 12)
// [Xn, #imm]! PRE_INDEXED (writeback before)
// [Xn], #imm POST_INDEXED (writeback after)
// [Xn, Xm{, LSL #s}] REG_OFFSET (shift = log2(size) when present)
// [Xn, Wm, SXTW|UXTW|SXTX #s] EXT_REG_OFFSET
// label LITERAL (PC-relative for LDR literal)
//
// `Shift_Type` and `Extend` enumerate the shifter/extender flavours that
// data-processing register and memory operand encodings need.
Operand_Kind :: enum u8 {
NONE,
REGISTER,
IMMEDIATE,
MEMORY,
RELATIVE,
SHIFTED_REG, // X reg + shift type + shift amount
EXTENDED_REG, // X/W reg + extend + amount
COND, // 4-bit condition code (EQ/NE/.../AL/NV)
ZA_SLICE, // `za0h.b[w12, 0]` -- a row or column of an SME tile
}
// One slice of an SME accumulator tile: which tile, taken along the rows (h)
// or the columns (v), addressed by one of W12..W15 plus a fixed offset.
ZA_Slice :: bit_field u32 {
tile: u8 | 4,
vertical: bool | 1,
ws: u8 | 2, // 0..3, meaning W12..W15
offset: u8 | 4,
elem: u8 | 5, // the ZSHAPE_* code the tile is viewed at
}
Shift_Type :: enum u8 {
LSL = 0,
LSR = 1,
ASR = 2,
ROR = 3,
}
Extend :: enum u8 {
UXTB = 0,
UXTH = 1,
UXTW = 2,
UXTX = 3,
SXTB = 4,
SXTH = 5,
SXTW = 6,
SXTX = 7,
}
Address_Mode :: enum u8 {
OFFSET, // [Xn, #imm] (imm may be 0)
PRE_INDEXED, // [Xn, #imm]!
POST_INDEXED, // [Xn], #imm
REG_OFFSET, // [Xn, Xm{, LSL #s}]
EXT_REG_OFFSET, // [Xn, Wm, SXTW|UXTW|SXTX #s]
LITERAL, // PC-rel target (LDR literal)
}
// Memory operand packed into one word: base + optional index + signed disp +
// addressing metadata. Index is `NONE` for non-register-offset modes.
//
// A bit_field rather than a struct because this sits in every Operand, so its
// width is multiplied by four in every Instruction. Field syntax is unchanged
// (`m.base`, `m.disp`) and composite literals still work, so this is invisible
// to callers.
//
// Widths: registers get 16 bits. `Register` itself is a u32, but every class
// legal in an address lives entirely in its low 16 bits -- only a system
// register (class REG_SYS) carries field bits above them, and one is never a
// valid base or index -- so the truncation is lossless and the NONE sentinel
// (0xFFFF) round-trips. That leaves 23 bits for `disp` (+/-4.19M) against a
// worst case of 65,520 -- LDR Q, [Xn, #imm12*16] -- the largest displacement
// any A64 addressing mode can encode, so there is ~64x headroom.
Memory :: bit_field u64 {
base: Register | 16,
index: Register | 16, // NONE for OFFSET/PRE/POST/LITERAL
disp: i32 | 23,
extend: Extend | 3, // for EXT_REG_OFFSET; UXTX otherwise
shift: u8 | 3, // 0..4 for register-offset / extended
mode: Address_Mode | 3,
// Full: 16 + 16 + 23 + 3 + 3 + 3 = 64.
}
#assert(size_of(Memory) == 8)
Shifted_Reg :: struct #packed {
reg: Register, // 4
type: Shift_Type, // 1
amount: u8, // 1 (0..63 for 64-bit; 0..31 for 32-bit)
}
#assert(size_of(Shifted_Reg) == 6)
Extended_Reg :: struct #packed {
reg: Register, // 4
extend: Extend, // 1
amount: u8, // 1 (0..4)
}
#assert(size_of(Extended_Reg) == 6)
// 11-byte tagged operand. The union holds whichever payload matches `kind`.
Operand :: struct #packed {
using _: struct #raw_union #packed {
reg: Register, // 4
mem: Memory, // 8
immediate: i64, // 8
relative: i64, // 8
shifted: Shifted_Reg, // 6
extended: Extended_Reg, // 6
cond: u8, // 1
za: ZA_Slice, // 4
}, // 8 total -- the largest member wins
kind: Operand_Kind, // 1
size: u8, // 1 -- carried width info; meaning varies
// How many consecutive registers the syntax writes as a list, starting at
// `reg`: `{v0.16b, v1.16b}` is 2. Zero means the operand is a plain
// register. The count belongs to the instruction form rather than to the
// caller -- LD2 always names two -- so it comes from the encoding.
list_count: u8, // 1
}
#assert(size_of(Operand) == 11)
// -----------------------------------------------------------------------------
// Constructors -- generic
// -----------------------------------------------------------------------------
@(require_results)
op_reg :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 4}
}
@(require_results)
op_imm :: #force_inline proc "contextless" (v: i64, size: u8 = 4) -> Operand {
return Operand{immediate = v, kind = .IMMEDIATE, size = size}
}
@(require_results)
op_label :: #force_inline proc "contextless" (label_id: u32, size: u8 = 4) -> Operand {
return Operand{relative = i64(label_id), kind = .RELATIVE, size = size}
}
@(require_results)
op_rel_offset :: #force_inline proc "contextless" (off: i64) -> Operand {
return Operand{relative = off, kind = .RELATIVE, size = 4}
}
@(require_results)
op_mem :: #force_inline proc "contextless" (m: Memory) -> Operand {
return Operand{mem = m, kind = .MEMORY, size = 4}
}
@(require_results)
op_shifted :: #force_inline proc "contextless" (r: Register, type: Shift_Type, amount: u8) -> Operand {
return Operand{shifted = Shifted_Reg{reg = r, type = type, amount = amount}, kind = .SHIFTED_REG, size = 4}
}
@(require_results)
op_extended :: #force_inline proc "contextless" (r: Register, ext: Extend, amount: u8) -> Operand {
return Operand{extended = Extended_Reg{reg = r, extend = ext, amount = amount}, kind = .EXTENDED_REG, size = 4}
}
@(require_results)
op_cond :: #force_inline proc "contextless" (c: Cond) -> Operand {
return Operand{cond = u8(c), kind = .COND, size = 1}
}
// A vector lane index. It is a plain immediate in the encoding, but it prints
// glued to the register it indexes (`v2.s[3]`) rather than as a separate
// operand, so it is marked to tell it apart from an immediate that really is
// one -- EXT's byte index, for instance, is written `#3`.
LANE_INDEX :: u8(0xFF)
// SVE writes its element-count pattern by name (`vl8`, `mul3`, `all`) and its
// multiplier as `mul #N`, so both need telling apart from a plain immediate.
SVE_PATTERN_IMM :: u8(0xFE)
SVE_MUL_IMM :: u8(0xFD)
// ZERO's operand is an 8-bit mask, one bit per .d tile, written as the list of
// the largest tiles that exactly cover it: a .s tile is two .d tiles four
// apart, a .h tile is four two apart, and the single .b tile is all eight.
ZA_TILE_MASK :: u8(0xFC)
// The 32 SVE element-count patterns; the gaps are reserved and print as a
// bare number.
@(rodata)
SVE_PATTERN_NAMES := [32]string{
"pow2", "vl1", "vl2", "vl3", "vl4", "vl5", "vl6", "vl7",
"vl8", "vl16", "vl32", "vl64", "vl128", "vl256", "", "",
"", "", "", "", "", "", "", "",
"", "", "", "", "", "mul4", "mul3", "all",
}
@(require_results)
op_lane_index :: #force_inline proc "contextless" (index: i64) -> Operand {
return Operand{immediate = index, kind = .IMMEDIATE, size = LANE_INDEX}
}
// A system register operand -- op_reg with the intent in the name, kept so
// MRS/MSR call sites read as what they are.
@(require_results)
op_sysreg :: #force_inline proc "contextless" (sr: Register) -> Operand {
return op_reg(sr)
}
@(require_results)
op_za_slice :: #force_inline proc "contextless" (tile, ws, offset, elem: u8, vertical := false) -> Operand {
return Operand{
za = ZA_Slice{tile = tile, vertical = vertical, ws = ws, offset = offset, elem = elem},
kind = .ZA_SLICE, size = elem,
}
}
// -----------------------------------------------------------------------------
// SVE Z-register builders -- encode the element arrangement in op.size
// (B=1, H=2, S=4, D=8). Matcher uses op.size to disambiguate the right
// table form when multiple element sizes share a base mnemonic.
// -----------------------------------------------------------------------------
@(require_results)
op_z_b :: #force_inline proc "contextless" (n: u8) -> Operand {
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 1}
}
@(require_results)
op_z_h :: #force_inline proc "contextless" (n: u8) -> Operand {
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 2}
}
@(require_results)
op_z_s :: #force_inline proc "contextless" (n: u8) -> Operand {
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 4}
}
@(require_results)
op_z_d :: #force_inline proc "contextless" (n: u8) -> Operand {
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 8}
}
// SVE packs an element size and a shift amount into one field, `tszh:tszl:imm3`,
// as `V = 2*esize - shift`. Because the shift is in [1, esize], V lands in
// [esize, 2*esize), so the four element sizes occupy disjoint ranges and the
// position of the highest set bit names the size:
//
// .b V in [ 8, 15] .s V in [32, 63]
// .h V in [16, 31] .d V in [64, 127]
//
// Encoder and decoder both need this, and it is written once here so the two
// cannot drift apart.
// The element width, in bits, that a packed tsz value names.
@(require_results)
sve_tsz_esize :: #force_inline proc "contextless" (v: u32) -> u32 {
switch {
case v >= 64: return 64
case v >= 32: return 32
case v >= 16: return 16
}
return 8
}
// The shift amount a packed tsz value names.
@(require_results)
sve_tsz_shift :: #force_inline proc "contextless" (v: u32) -> u32 {
return 2 * sve_tsz_esize(v) - v
}
// The packed tsz value for an element width and shift.
@(require_results)
sve_tsz_pack :: #force_inline proc "contextless" (esize, shift: u32) -> u32 {
return (2 * esize - shift) & 0x7F
}
// The SVE element-width code (the `size` an operand carries) for a width in
// bits: .b = 1, .h = 2, .s = 4, .d = 8.
@(require_results)
sve_esize_code :: #force_inline proc "contextless" (esize: u32) -> u8 {
return u8(esize / 8)
}
// A predicate register's governing qualifier, carried in Operand.size and read
// only when the register's class is REG_P. SVE writes it as a suffix -- `p0/z`
// zeroes the inactive lanes, `p0/m` leaves them alone -- and an assembler will
// not take the instruction without it where the form calls for one.
PQUAL_ZERO :: u8(1)
PQUAL_MERGE :: u8(2)
PQUAL_NONE :: u8(4)
// A predicate that is an instruction's DESTINATION is written with an element
// size instead -- `cmpge p0.b, p1/z, ...` -- so those codes have to live apart
// from the qualifiers above, and apart from the neutral 4 a plain op_reg gives.
PSHAPE_B :: u8(20)
PSHAPE_H :: u8(21)
PSHAPE_S :: u8(22)
PSHAPE_D :: u8(23)
// Arrangement codes carried in Operand.size. Element views are odd and
// arrangements are multiples of 8, so the two can never be confused; 4 is the
// neutral "no vector shape" value every scalar class uses.
VSHAPE_NONE :: u8(4)
VSHAPE_8B :: u8(8)
VSHAPE_16B :: u8(16)
VSHAPE_4H :: u8(24)
VSHAPE_8H :: u8(32)
VSHAPE_2S :: u8(40)
VSHAPE_4S :: u8(48)
VSHAPE_1D :: u8(56)
VSHAPE_2D :: u8(64)
VSHAPE_1Q :: u8(72)
VSHAPE_ELEM_B :: u8(1)
VSHAPE_ELEM_H :: u8(3)
VSHAPE_ELEM_S :: u8(5)
VSHAPE_ELEM_D :: u8(7)
// SVE element-width codes carried in Operand.size for a Z register.
ZSHAPE_B :: u8(1)
ZSHAPE_H :: u8(2)
ZSHAPE_S :: u8(4)
ZSHAPE_D :: u8(8)
ZSHAPE_Q :: u8(16)
// -----------------------------------------------------------------------------
// NEON V-register arrangement builders. These take the register the caller
// actually has, not its number: rebuilding one from `reg_hw` would relabel an
// X register as a V register, and the matcher -- which checks reg_class --
// would never get to reject it. Passing a non-V register here now simply
// matches no form, and encode reports it.
//
// op.size encodes lanes*elem-bytes:
// .8B = 8 .16B = 16
// .4H = 24 .8H = 32
// .2S = 40 .4S = 48
// .1D = 56 .2D = 64
// (Encoded so that no two arrangements collide and so the value is easy
// to inspect.)
// -----------------------------------------------------------------------------
@(require_results)
op_v_8b :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 8}
}
@(require_results)
op_v_16b :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 16}
}
@(require_results)
op_v_4h :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 24}
}
@(require_results)
op_v_8h :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 32}
}
@(require_results)
op_v_2s :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 40}
}
@(require_results)
op_v_4s :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 48}
}
@(require_results)
op_v_1d :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 56}
}
@(require_results)
op_v_2d :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 64}
}
// .1q breaks the lanes*elem-bytes rule the others follow (1*16 would collide
// with 16B), so it gets the next free multiple of 8.
@(require_results)
op_v_1q :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 72}
}
// A run of `count` consecutive registers written as a list, `{v0.16b, v1.16b}`.
// `shape` is one of the VSHAPE_* codes.
@(require_results)
op_v_list :: #force_inline proc "contextless" (first: Register, shape, count: u8) -> Operand {
return Operand{reg = first, kind = .REGISTER, size = shape, list_count = count}
}
// Element-indexed V views (V0.B[i]/.H[i]/.S[i]/.D[i]). The element size rides
// in op.size so the matcher can disambiguate DUP/INS forms; the lane index is
// a separate immediate operand.
//
// The codes are ODD (1/3/5/7) on purpose: arrangement operands above use
// multiples of 8, so a size can never mean both. They used to be 1/2/4/8,
// which made an element-D view indistinguishable from an 8B arrangement --
// the printer cannot tell `.d` from `.8b` if both are size 8.
@(require_results)
op_v_elem_b :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 1}
}
@(require_results)
op_v_elem_h :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 3}
}
@(require_results)
op_v_elem_s :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 5}
}
@(require_results)
op_v_elem_d :: #force_inline proc "contextless" (r: Register) -> Operand {
return Operand{reg = r, kind = .REGISTER, size = 7}
}
// -----------------------------------------------------------------------------
// Memory constructors (one per addressing mode)
// -----------------------------------------------------------------------------
@(require_results)
mem_offset :: #force_inline proc "contextless" (base: Register, disp: i32 = 0) -> Memory {
return Memory{base = base, index = NONE, disp = disp, mode = .OFFSET}
}
@(require_results)
mem_pre :: #force_inline proc "contextless" (base: Register, disp: i32) -> Memory {
return Memory{base = base, index = NONE, disp = disp, mode = .PRE_INDEXED}
}
@(require_results)
mem_post :: #force_inline proc "contextless" (base: Register, disp: i32) -> Memory {
return Memory{base = base, index = NONE, disp = disp, mode = .POST_INDEXED}
}
@(require_results)
mem_reg :: #force_inline proc "contextless" (base, index: Register, shift_amount: u8 = 0) -> Memory {
return Memory{base = base, index = index, mode = .REG_OFFSET, shift = shift_amount, extend = .UXTX}
}
@(require_results)
mem_ext :: #force_inline proc "contextless" (base, index: Register, ext: Extend, shift_amount: u8 = 0) -> Memory {
return Memory{base = base, index = index, mode = .EXT_REG_OFFSET, extend = ext, shift = shift_amount}
}
// -----------------------------------------------------------------------------
// Condition codes
// -----------------------------------------------------------------------------
Cond :: enum u8 {
EQ = 0x0,
NE = 0x1,
CS = 0x2, // unsigned higher or same (alias HS)
CC = 0x3, // unsigned lower (alias LO)
MI = 0x4,
PL = 0x5,
VS = 0x6,
VC = 0x7,
HI = 0x8,
LS = 0x9,
GE = 0xA,
LT = 0xB,
GT = 0xC,
LE = 0xD,
AL = 0xE,
NV = 0xF,
}
// Architectural aliases for the two carry-style conditions.
COND_HS :: Cond.CS
COND_LO :: Cond.CC