mirror of
https://github.com/odin-lang/Odin.git
synced 2026-09-02 02:03:35 +00:00
Splitting B_COND left three places still describing the old model. The generated builders were already right -- inst_b_le(label), inst_bc_ne(label), one per condition, all 32 verified to encode and round-trip -- but the hand-written scaffolding around them was not. verify_against_llvm normalised our mnemonic by truncating at the first underscore, which turned B_COND into "b". LLVM prints b.eq/b.ne/..., so the tool carried 32 alias rows pairing "b" with each of them to stop the mismatch being reported. That truncation now collapses all sixteen B_* onto "b" and makes every condition compare equal to every other -- the check would pass whatever the table said. Keep the condition instead (B_LE -> "b.le") and the 32 alias rows are unnecessary; they are gone. specgen's canonicalizer kept B_COND and BC_COND off its rename path by name. Those names no longer exist, so replace the entry with a rule that matches the shape (BC?_%u%u), which is what the intent was. And the note in instructions.odin still pointed at inst_b_cond. It now says what is actually true: a conditional branch is one builder per condition because the condition is part of the mnemonic, while the select/compare family -- CSEL, CSINC, CSINV, CSNEG, CCMP, CCMN, FCSEL -- really does take a condition operand and keeps one. specgen still re-derives its 1130 forms from llvm-mc and finds every one already present, all suites pass, and all 13 packages build. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
128 lines
5.4 KiB
Odin
128 lines
5.4 KiB
Odin
// rexcode · Brendan Punsky (dotbmp@github), original author
|
|
|
|
package rexcode_arm64
|
|
|
|
// =============================================================================
|
|
// INSTRUCTION
|
|
// =============================================================================
|
|
|
|
Instruction_Flags :: bit_field u8 {
|
|
_: u8 | 8,
|
|
}
|
|
|
|
// Sized and aligned to a cache line, deliberately.
|
|
//
|
|
// The payload is 45 bytes -- shrinking Operand to 10 got it there -- but
|
|
// leaving the struct at 48 was measurably worse than padding it back out.
|
|
// With `#packed` the struct aligns to 1, so a 48-byte stride straddles a line
|
|
// boundary 75% of the time and a 64-byte one straddled 100% of the time (the
|
|
// heap base is not line-aligned either). Decode writes whole Instructions, and
|
|
// unaligned stores cost enough that on an i7-9750H this layout decodes ~21%
|
|
// faster than the 48-byte packed one -- while writing MORE bytes. The gap
|
|
// holds even when the whole array is L1-resident, so it is split-store cost at
|
|
// the store ports, not cache-line fetches.
|
|
//
|
|
// Encode is within 1% and a pure read traversal is ~9% slower at working sets
|
|
// past L2, both of which the decode win dwarfs for real workloads.
|
|
//
|
|
// The 19 spare bytes are free: they cost nothing over a 48-byte struct that
|
|
// straddles, and new fields land in them without changing the layout.
|
|
Instruction :: struct #align(64) {
|
|
ops: [4]Operand `fmt:"v,operand_count"`, // 4 * size_of(Operand) = 40
|
|
mnemonic: Mnemonic, // 2
|
|
operand_count: u8, // 1
|
|
flags: Instruction_Flags, // 1
|
|
length: u8, // 1 -- always 4
|
|
_: [19]u8,
|
|
}
|
|
#assert(size_of(Instruction) == 64)
|
|
#assert(align_of(Instruction) == 64)
|
|
|
|
// =============================================================================
|
|
// Builders -- the most common shapes; less-common forms can be built
|
|
// inline by the caller using the Instruction struct directly.
|
|
// =============================================================================
|
|
|
|
@(require_results)
|
|
inst_none :: #force_inline proc "contextless" (m: Mnemonic) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 0, length = 4}
|
|
}
|
|
|
|
// Single-register (e.g. BR, BLR).
|
|
@(require_results)
|
|
inst_r :: #force_inline proc "contextless" (m: Mnemonic, r: Register) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 1, length = 4,
|
|
ops = {op_reg(r), {}, {}, {}}}
|
|
}
|
|
|
|
// 2-register (e.g. CLZ, RBIT).
|
|
@(require_results)
|
|
inst_r_r :: #force_inline proc "contextless" (m: Mnemonic, rd, rn: Register) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 2, length = 4,
|
|
ops = {op_reg(rd), op_reg(rn), {}, {}}}
|
|
}
|
|
|
|
// 3-register (e.g. ADD shifted, MUL, UDIV, ASRV).
|
|
@(require_results)
|
|
inst_r_r_r :: #force_inline proc "contextless" (m: Mnemonic, rd, rn, rm: Register) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 3, length = 4,
|
|
ops = {op_reg(rd), op_reg(rn), op_reg(rm), {}}}
|
|
}
|
|
|
|
// 4-register R4-type (MADD, MSUB, SMADDL, ...).
|
|
@(require_results)
|
|
inst_r_r_r_r :: #force_inline proc "contextless" (m: Mnemonic, rd, rn, rm, ra: Register) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 4, length = 4,
|
|
ops = {op_reg(rd), op_reg(rn), op_reg(rm), op_reg(ra)}}
|
|
}
|
|
|
|
// 2-register + immediate (e.g. ADD imm).
|
|
@(require_results)
|
|
inst_r_r_i :: #force_inline proc "contextless" (m: Mnemonic, rd, rn: Register, imm: i64) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 3, length = 4,
|
|
ops = {op_reg(rd), op_reg(rn), op_imm(imm), {}}}
|
|
}
|
|
|
|
// 1-register + immediate (e.g. MOVZ).
|
|
@(require_results)
|
|
inst_r_i :: #force_inline proc "contextless" (m: Mnemonic, rd: Register, imm: i64) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 2, length = 4,
|
|
ops = {op_reg(rd), op_imm(imm), {}, {}}}
|
|
}
|
|
|
|
// MOVZ/MOVN/MOVK with explicit hw shift (0/16/32/48).
|
|
@(require_results)
|
|
inst_mov_imm :: #force_inline proc "contextless" (m: Mnemonic, rd: Register, imm: i64, hw: u8) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 3, length = 4,
|
|
ops = {op_reg(rd), op_imm(imm), op_imm(i64(hw), 1), {}}}
|
|
}
|
|
|
|
// Load/store register: Rt + memory.
|
|
@(require_results)
|
|
inst_ldst :: #force_inline proc "contextless" (m: Mnemonic, rt: Register, mm: Memory) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 2, length = 4,
|
|
ops = {op_reg(rt), op_mem(mm), {}, {}}}
|
|
}
|
|
|
|
// Load/store pair: Rt, Rt2, memory.
|
|
@(require_results)
|
|
inst_ldp_stp :: #force_inline proc "contextless" (m: Mnemonic, rt, rt2: Register, mm: Memory) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 3, length = 4,
|
|
ops = {op_reg(rt), op_reg(rt2), op_mem(mm), {}}}
|
|
}
|
|
|
|
// PC-relative branch (B, BL).
|
|
@(require_results)
|
|
inst_branch :: #force_inline proc "contextless" (m: Mnemonic, label_id: u32) -> Instruction {
|
|
return Instruction{mnemonic = m, operand_count = 1, length = 4,
|
|
ops = {op_label(label_id, 4), {}, {}, {}}}
|
|
}
|
|
|
|
// NOTE: the conditional branches, inst_cbz (+cbnz), inst_tbz (+tbnz) and
|
|
// inst_csel (+csinc/csinv/csneg) are generated per-mnemonic in
|
|
// mnemonic_builders.odin, so the generator owns those names. A conditional
|
|
// branch is one builder per condition -- inst_b_le(label), inst_bc_ne(label)
|
|
// -- because the condition is part of the mnemonic, not an operand; the
|
|
// condition-operand builders are the select/compare family, which really do
|
|
// take one (inst_csinc(rd, rn, rm, cond)).
|