mirror of
https://github.com/odin-lang/Odin.git
synced 2026-09-02 02:03:35 +00:00
Register widens to a u32: hw number in bits 0-4, class byte in bits 8-15 -- bit-identical to the old u16 layout -- and, for the new REG_SYS class only, the 15-bit MRS/MSR field in bits 16-30. System_Register, its Operand_Kind, and the union's sysreg member are gone; a system register is now a plain .REGISTER operand distinguished by class, so it flows through matching, packing, and printing like any other register. Memory is untouched: every class legal in an address still lives entirely in the low 16 bits of the u32, so its 16-bit register slots stay lossless and NONE round-trips (verified: Odin bit_fields zero-extend on read and reject overflowing constants at compile time). Operand stays 11 bytes -- the union's largest member is still 8 -- and Instruction stays exactly 64. op_sysreg survives as an op_reg alias so MRS/MSR call sites read as what they are, and the sysreg constants keep their field value in their name: NZCV is now Register(0x5A10_1000) where it was System_Register(0x5A10). New pipeline test: every SYSREG_NAMES entry round-trips encode -> decode -> print byte-exactly, with the expected word derived from the table value. All 330 table + 134 pipeline checks pass; benchmarks show encode ~3% faster, decode and print at parity. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01AFeLCDKi5kRMtHrskUaRfw
462 lines
17 KiB
Odin
462 lines
17 KiB
Odin
// rexcode · Brendan Punsky (dotbmp@github), original author
|
|
|
|
package rexcode_arm64
|
|
|
|
// =============================================================================
|
|
// AArch64 OPERANDS
|
|
// =============================================================================
|
|
//
|
|
// AArch64 has a rich addressing repertoire:
|
|
//
|
|
// [Xn] OFFSET with imm=0
|
|
// [Xn, #imm] OFFSET (signed 9 or unsigned scaled 12)
|
|
// [Xn, #imm]! PRE_INDEXED (writeback before)
|
|
// [Xn], #imm POST_INDEXED (writeback after)
|
|
// [Xn, Xm{, LSL #s}] REG_OFFSET (shift = log2(size) when present)
|
|
// [Xn, Wm, SXTW|UXTW|SXTX #s] EXT_REG_OFFSET
|
|
// label LITERAL (PC-relative for LDR literal)
|
|
//
|
|
// `Shift_Type` and `Extend` enumerate the shifter/extender flavours that
|
|
// data-processing register and memory operand encodings need.
|
|
|
|
Operand_Kind :: enum u8 {
|
|
NONE,
|
|
REGISTER,
|
|
IMMEDIATE,
|
|
MEMORY,
|
|
RELATIVE,
|
|
SHIFTED_REG, // X reg + shift type + shift amount
|
|
EXTENDED_REG, // X/W reg + extend + amount
|
|
COND, // 4-bit condition code (EQ/NE/.../AL/NV)
|
|
ZA_SLICE, // `za0h.b[w12, 0]` -- a row or column of an SME tile
|
|
}
|
|
|
|
// One slice of an SME accumulator tile: which tile, taken along the rows (h)
|
|
// or the columns (v), addressed by one of W12..W15 plus a fixed offset.
|
|
ZA_Slice :: bit_field u32 {
|
|
tile: u8 | 4,
|
|
vertical: bool | 1,
|
|
ws: u8 | 2, // 0..3, meaning W12..W15
|
|
offset: u8 | 4,
|
|
elem: u8 | 5, // the ZSHAPE_* code the tile is viewed at
|
|
}
|
|
|
|
Shift_Type :: enum u8 {
|
|
LSL = 0,
|
|
LSR = 1,
|
|
ASR = 2,
|
|
ROR = 3,
|
|
}
|
|
|
|
Extend :: enum u8 {
|
|
UXTB = 0,
|
|
UXTH = 1,
|
|
UXTW = 2,
|
|
UXTX = 3,
|
|
SXTB = 4,
|
|
SXTH = 5,
|
|
SXTW = 6,
|
|
SXTX = 7,
|
|
}
|
|
|
|
Address_Mode :: enum u8 {
|
|
OFFSET, // [Xn, #imm] (imm may be 0)
|
|
PRE_INDEXED, // [Xn, #imm]!
|
|
POST_INDEXED, // [Xn], #imm
|
|
REG_OFFSET, // [Xn, Xm{, LSL #s}]
|
|
EXT_REG_OFFSET, // [Xn, Wm, SXTW|UXTW|SXTX #s]
|
|
LITERAL, // PC-rel target (LDR literal)
|
|
}
|
|
|
|
// Memory operand packed into one word: base + optional index + signed disp +
|
|
// addressing metadata. Index is `NONE` for non-register-offset modes.
|
|
//
|
|
// A bit_field rather than a struct because this sits in every Operand, so its
|
|
// width is multiplied by four in every Instruction. Field syntax is unchanged
|
|
// (`m.base`, `m.disp`) and composite literals still work, so this is invisible
|
|
// to callers.
|
|
//
|
|
// Widths: registers get 16 bits. `Register` itself is a u32, but every class
|
|
// legal in an address lives entirely in its low 16 bits -- only a system
|
|
// register (class REG_SYS) carries field bits above them, and one is never a
|
|
// valid base or index -- so the truncation is lossless and the NONE sentinel
|
|
// (0xFFFF) round-trips. That leaves 23 bits for `disp` (+/-4.19M) against a
|
|
// worst case of 65,520 -- LDR Q, [Xn, #imm12*16] -- the largest displacement
|
|
// any A64 addressing mode can encode, so there is ~64x headroom.
|
|
Memory :: bit_field u64 {
|
|
base: Register | 16,
|
|
index: Register | 16, // NONE for OFFSET/PRE/POST/LITERAL
|
|
disp: i32 | 23,
|
|
extend: Extend | 3, // for EXT_REG_OFFSET; UXTX otherwise
|
|
shift: u8 | 3, // 0..4 for register-offset / extended
|
|
mode: Address_Mode | 3,
|
|
// Full: 16 + 16 + 23 + 3 + 3 + 3 = 64.
|
|
}
|
|
#assert(size_of(Memory) == 8)
|
|
|
|
Shifted_Reg :: struct #packed {
|
|
reg: Register, // 4
|
|
type: Shift_Type, // 1
|
|
amount: u8, // 1 (0..63 for 64-bit; 0..31 for 32-bit)
|
|
}
|
|
#assert(size_of(Shifted_Reg) == 6)
|
|
|
|
Extended_Reg :: struct #packed {
|
|
reg: Register, // 4
|
|
extend: Extend, // 1
|
|
amount: u8, // 1 (0..4)
|
|
}
|
|
#assert(size_of(Extended_Reg) == 6)
|
|
|
|
// 11-byte tagged operand. The union holds whichever payload matches `kind`.
|
|
Operand :: struct #packed {
|
|
using _: struct #raw_union #packed {
|
|
reg: Register, // 4
|
|
mem: Memory, // 8
|
|
immediate: i64, // 8
|
|
relative: i64, // 8
|
|
shifted: Shifted_Reg, // 6
|
|
extended: Extended_Reg, // 6
|
|
cond: u8, // 1
|
|
za: ZA_Slice, // 4
|
|
}, // 8 total -- the largest member wins
|
|
kind: Operand_Kind, // 1
|
|
size: u8, // 1 -- carried width info; meaning varies
|
|
// How many consecutive registers the syntax writes as a list, starting at
|
|
// `reg`: `{v0.16b, v1.16b}` is 2. Zero means the operand is a plain
|
|
// register. The count belongs to the instruction form rather than to the
|
|
// caller -- LD2 always names two -- so it comes from the encoding.
|
|
list_count: u8, // 1
|
|
}
|
|
#assert(size_of(Operand) == 11)
|
|
|
|
// -----------------------------------------------------------------------------
|
|
// Constructors -- generic
|
|
// -----------------------------------------------------------------------------
|
|
|
|
@(require_results)
|
|
op_reg :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 4}
|
|
}
|
|
@(require_results)
|
|
op_imm :: #force_inline proc "contextless" (v: i64, size: u8 = 4) -> Operand {
|
|
return Operand{immediate = v, kind = .IMMEDIATE, size = size}
|
|
}
|
|
@(require_results)
|
|
op_label :: #force_inline proc "contextless" (label_id: u32, size: u8 = 4) -> Operand {
|
|
return Operand{relative = i64(label_id), kind = .RELATIVE, size = size}
|
|
}
|
|
@(require_results)
|
|
op_rel_offset :: #force_inline proc "contextless" (off: i64) -> Operand {
|
|
return Operand{relative = off, kind = .RELATIVE, size = 4}
|
|
}
|
|
|
|
@(require_results)
|
|
op_mem :: #force_inline proc "contextless" (m: Memory) -> Operand {
|
|
return Operand{mem = m, kind = .MEMORY, size = 4}
|
|
}
|
|
|
|
@(require_results)
|
|
op_shifted :: #force_inline proc "contextless" (r: Register, type: Shift_Type, amount: u8) -> Operand {
|
|
return Operand{shifted = Shifted_Reg{reg = r, type = type, amount = amount}, kind = .SHIFTED_REG, size = 4}
|
|
}
|
|
|
|
@(require_results)
|
|
op_extended :: #force_inline proc "contextless" (r: Register, ext: Extend, amount: u8) -> Operand {
|
|
return Operand{extended = Extended_Reg{reg = r, extend = ext, amount = amount}, kind = .EXTENDED_REG, size = 4}
|
|
}
|
|
|
|
@(require_results)
|
|
op_cond :: #force_inline proc "contextless" (c: Cond) -> Operand {
|
|
return Operand{cond = u8(c), kind = .COND, size = 1}
|
|
}
|
|
|
|
// A vector lane index. It is a plain immediate in the encoding, but it prints
|
|
// glued to the register it indexes (`v2.s[3]`) rather than as a separate
|
|
// operand, so it is marked to tell it apart from an immediate that really is
|
|
// one -- EXT's byte index, for instance, is written `#3`.
|
|
LANE_INDEX :: u8(0xFF)
|
|
|
|
// SVE writes its element-count pattern by name (`vl8`, `mul3`, `all`) and its
|
|
// multiplier as `mul #N`, so both need telling apart from a plain immediate.
|
|
SVE_PATTERN_IMM :: u8(0xFE)
|
|
SVE_MUL_IMM :: u8(0xFD)
|
|
|
|
// ZERO's operand is an 8-bit mask, one bit per .d tile, written as the list of
|
|
// the largest tiles that exactly cover it: a .s tile is two .d tiles four
|
|
// apart, a .h tile is four two apart, and the single .b tile is all eight.
|
|
ZA_TILE_MASK :: u8(0xFC)
|
|
|
|
// The 32 SVE element-count patterns; the gaps are reserved and print as a
|
|
// bare number.
|
|
@(rodata)
|
|
SVE_PATTERN_NAMES := [32]string{
|
|
"pow2", "vl1", "vl2", "vl3", "vl4", "vl5", "vl6", "vl7",
|
|
"vl8", "vl16", "vl32", "vl64", "vl128", "vl256", "", "",
|
|
"", "", "", "", "", "", "", "",
|
|
"", "", "", "", "", "mul4", "mul3", "all",
|
|
}
|
|
|
|
@(require_results)
|
|
op_lane_index :: #force_inline proc "contextless" (index: i64) -> Operand {
|
|
return Operand{immediate = index, kind = .IMMEDIATE, size = LANE_INDEX}
|
|
}
|
|
|
|
// A system register operand -- op_reg with the intent in the name, kept so
|
|
// MRS/MSR call sites read as what they are.
|
|
@(require_results)
|
|
op_sysreg :: #force_inline proc "contextless" (sr: Register) -> Operand {
|
|
return op_reg(sr)
|
|
}
|
|
|
|
@(require_results)
|
|
op_za_slice :: #force_inline proc "contextless" (tile, ws, offset, elem: u8, vertical := false) -> Operand {
|
|
return Operand{
|
|
za = ZA_Slice{tile = tile, vertical = vertical, ws = ws, offset = offset, elem = elem},
|
|
kind = .ZA_SLICE, size = elem,
|
|
}
|
|
}
|
|
|
|
// -----------------------------------------------------------------------------
|
|
// SVE Z-register builders -- encode the element arrangement in op.size
|
|
// (B=1, H=2, S=4, D=8). Matcher uses op.size to disambiguate the right
|
|
// table form when multiple element sizes share a base mnemonic.
|
|
// -----------------------------------------------------------------------------
|
|
|
|
@(require_results)
|
|
op_z_b :: #force_inline proc "contextless" (n: u8) -> Operand {
|
|
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 1}
|
|
}
|
|
@(require_results)
|
|
op_z_h :: #force_inline proc "contextless" (n: u8) -> Operand {
|
|
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 2}
|
|
}
|
|
@(require_results)
|
|
op_z_s :: #force_inline proc "contextless" (n: u8) -> Operand {
|
|
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 4}
|
|
}
|
|
@(require_results)
|
|
op_z_d :: #force_inline proc "contextless" (n: u8) -> Operand {
|
|
return Operand{reg = Register(REG_Z | u16(n & 0x1F)), kind = .REGISTER, size = 8}
|
|
}
|
|
|
|
// SVE packs an element size and a shift amount into one field, `tszh:tszl:imm3`,
|
|
// as `V = 2*esize - shift`. Because the shift is in [1, esize], V lands in
|
|
// [esize, 2*esize), so the four element sizes occupy disjoint ranges and the
|
|
// position of the highest set bit names the size:
|
|
//
|
|
// .b V in [ 8, 15] .s V in [32, 63]
|
|
// .h V in [16, 31] .d V in [64, 127]
|
|
//
|
|
// Encoder and decoder both need this, and it is written once here so the two
|
|
// cannot drift apart.
|
|
|
|
// The element width, in bits, that a packed tsz value names.
|
|
@(require_results)
|
|
sve_tsz_esize :: #force_inline proc "contextless" (v: u32) -> u32 {
|
|
switch {
|
|
case v >= 64: return 64
|
|
case v >= 32: return 32
|
|
case v >= 16: return 16
|
|
}
|
|
return 8
|
|
}
|
|
|
|
// The shift amount a packed tsz value names.
|
|
@(require_results)
|
|
sve_tsz_shift :: #force_inline proc "contextless" (v: u32) -> u32 {
|
|
return 2 * sve_tsz_esize(v) - v
|
|
}
|
|
|
|
// The packed tsz value for an element width and shift.
|
|
@(require_results)
|
|
sve_tsz_pack :: #force_inline proc "contextless" (esize, shift: u32) -> u32 {
|
|
return (2 * esize - shift) & 0x7F
|
|
}
|
|
|
|
// The SVE element-width code (the `size` an operand carries) for a width in
|
|
// bits: .b = 1, .h = 2, .s = 4, .d = 8.
|
|
@(require_results)
|
|
sve_esize_code :: #force_inline proc "contextless" (esize: u32) -> u8 {
|
|
return u8(esize / 8)
|
|
}
|
|
|
|
// A predicate register's governing qualifier, carried in Operand.size and read
|
|
// only when the register's class is REG_P. SVE writes it as a suffix -- `p0/z`
|
|
// zeroes the inactive lanes, `p0/m` leaves them alone -- and an assembler will
|
|
// not take the instruction without it where the form calls for one.
|
|
PQUAL_ZERO :: u8(1)
|
|
PQUAL_MERGE :: u8(2)
|
|
PQUAL_NONE :: u8(4)
|
|
|
|
// A predicate that is an instruction's DESTINATION is written with an element
|
|
// size instead -- `cmpge p0.b, p1/z, ...` -- so those codes have to live apart
|
|
// from the qualifiers above, and apart from the neutral 4 a plain op_reg gives.
|
|
PSHAPE_B :: u8(20)
|
|
PSHAPE_H :: u8(21)
|
|
PSHAPE_S :: u8(22)
|
|
PSHAPE_D :: u8(23)
|
|
|
|
// Arrangement codes carried in Operand.size. Element views are odd and
|
|
// arrangements are multiples of 8, so the two can never be confused; 4 is the
|
|
// neutral "no vector shape" value every scalar class uses.
|
|
VSHAPE_NONE :: u8(4)
|
|
VSHAPE_8B :: u8(8)
|
|
VSHAPE_16B :: u8(16)
|
|
VSHAPE_4H :: u8(24)
|
|
VSHAPE_8H :: u8(32)
|
|
VSHAPE_2S :: u8(40)
|
|
VSHAPE_4S :: u8(48)
|
|
VSHAPE_1D :: u8(56)
|
|
VSHAPE_2D :: u8(64)
|
|
VSHAPE_1Q :: u8(72)
|
|
VSHAPE_ELEM_B :: u8(1)
|
|
VSHAPE_ELEM_H :: u8(3)
|
|
VSHAPE_ELEM_S :: u8(5)
|
|
VSHAPE_ELEM_D :: u8(7)
|
|
|
|
// SVE element-width codes carried in Operand.size for a Z register.
|
|
ZSHAPE_B :: u8(1)
|
|
ZSHAPE_H :: u8(2)
|
|
ZSHAPE_S :: u8(4)
|
|
ZSHAPE_D :: u8(8)
|
|
ZSHAPE_Q :: u8(16)
|
|
|
|
// -----------------------------------------------------------------------------
|
|
// NEON V-register arrangement builders. These take the register the caller
|
|
// actually has, not its number: rebuilding one from `reg_hw` would relabel an
|
|
// X register as a V register, and the matcher -- which checks reg_class --
|
|
// would never get to reject it. Passing a non-V register here now simply
|
|
// matches no form, and encode reports it.
|
|
//
|
|
// op.size encodes lanes*elem-bytes:
|
|
// .8B = 8 .16B = 16
|
|
// .4H = 24 .8H = 32
|
|
// .2S = 40 .4S = 48
|
|
// .1D = 56 .2D = 64
|
|
// (Encoded so that no two arrangements collide and so the value is easy
|
|
// to inspect.)
|
|
// -----------------------------------------------------------------------------
|
|
|
|
@(require_results)
|
|
op_v_8b :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 8}
|
|
}
|
|
@(require_results)
|
|
op_v_16b :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 16}
|
|
}
|
|
@(require_results)
|
|
op_v_4h :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 24}
|
|
}
|
|
@(require_results)
|
|
op_v_8h :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 32}
|
|
}
|
|
@(require_results)
|
|
op_v_2s :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 40}
|
|
}
|
|
@(require_results)
|
|
op_v_4s :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 48}
|
|
}
|
|
@(require_results)
|
|
op_v_1d :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 56}
|
|
}
|
|
@(require_results)
|
|
op_v_2d :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 64}
|
|
}
|
|
// .1q breaks the lanes*elem-bytes rule the others follow (1*16 would collide
|
|
// with 16B), so it gets the next free multiple of 8.
|
|
@(require_results)
|
|
op_v_1q :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 72}
|
|
}
|
|
// A run of `count` consecutive registers written as a list, `{v0.16b, v1.16b}`.
|
|
// `shape` is one of the VSHAPE_* codes.
|
|
@(require_results)
|
|
op_v_list :: #force_inline proc "contextless" (first: Register, shape, count: u8) -> Operand {
|
|
return Operand{reg = first, kind = .REGISTER, size = shape, list_count = count}
|
|
}
|
|
|
|
// Element-indexed V views (V0.B[i]/.H[i]/.S[i]/.D[i]). The element size rides
|
|
// in op.size so the matcher can disambiguate DUP/INS forms; the lane index is
|
|
// a separate immediate operand.
|
|
//
|
|
// The codes are ODD (1/3/5/7) on purpose: arrangement operands above use
|
|
// multiples of 8, so a size can never mean both. They used to be 1/2/4/8,
|
|
// which made an element-D view indistinguishable from an 8B arrangement --
|
|
// the printer cannot tell `.d` from `.8b` if both are size 8.
|
|
@(require_results)
|
|
op_v_elem_b :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 1}
|
|
}
|
|
@(require_results)
|
|
op_v_elem_h :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 3}
|
|
}
|
|
@(require_results)
|
|
op_v_elem_s :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 5}
|
|
}
|
|
@(require_results)
|
|
op_v_elem_d :: #force_inline proc "contextless" (r: Register) -> Operand {
|
|
return Operand{reg = r, kind = .REGISTER, size = 7}
|
|
}
|
|
|
|
// -----------------------------------------------------------------------------
|
|
// Memory constructors (one per addressing mode)
|
|
// -----------------------------------------------------------------------------
|
|
|
|
@(require_results)
|
|
mem_offset :: #force_inline proc "contextless" (base: Register, disp: i32 = 0) -> Memory {
|
|
return Memory{base = base, index = NONE, disp = disp, mode = .OFFSET}
|
|
}
|
|
@(require_results)
|
|
mem_pre :: #force_inline proc "contextless" (base: Register, disp: i32) -> Memory {
|
|
return Memory{base = base, index = NONE, disp = disp, mode = .PRE_INDEXED}
|
|
}
|
|
@(require_results)
|
|
mem_post :: #force_inline proc "contextless" (base: Register, disp: i32) -> Memory {
|
|
return Memory{base = base, index = NONE, disp = disp, mode = .POST_INDEXED}
|
|
}
|
|
@(require_results)
|
|
mem_reg :: #force_inline proc "contextless" (base, index: Register, shift_amount: u8 = 0) -> Memory {
|
|
return Memory{base = base, index = index, mode = .REG_OFFSET, shift = shift_amount, extend = .UXTX}
|
|
}
|
|
@(require_results)
|
|
mem_ext :: #force_inline proc "contextless" (base, index: Register, ext: Extend, shift_amount: u8 = 0) -> Memory {
|
|
return Memory{base = base, index = index, mode = .EXT_REG_OFFSET, extend = ext, shift = shift_amount}
|
|
}
|
|
|
|
// -----------------------------------------------------------------------------
|
|
// Condition codes
|
|
// -----------------------------------------------------------------------------
|
|
|
|
Cond :: enum u8 {
|
|
EQ = 0x0,
|
|
NE = 0x1,
|
|
CS = 0x2, // unsigned higher or same (alias HS)
|
|
CC = 0x3, // unsigned lower (alias LO)
|
|
MI = 0x4,
|
|
PL = 0x5,
|
|
VS = 0x6,
|
|
VC = 0x7,
|
|
HI = 0x8,
|
|
LS = 0x9,
|
|
GE = 0xA,
|
|
LT = 0xB,
|
|
GT = 0xC,
|
|
LE = 0xD,
|
|
AL = 0xE,
|
|
NV = 0xF,
|
|
}
|
|
|
|
// Architectural aliases for the two carry-style conditions.
|
|
COND_HS :: Cond.CS
|
|
COND_LO :: Cond.CC
|