riscv64: read a flattened aggregate's members from their real offsets

This commit is contained in:
kalsprite
2026-08-14 19:15:11 -07:00
parent fa342bb540
commit ba708a2ad2
2 changed files with 183 additions and 5 deletions

View File

@@ -16,6 +16,14 @@ struct lbArgType {
i64 byval_alignment;
bool is_byval;
bool no_capture;
// For RiscV (Optional for others): A `cast_type` is normally applied by reinterpreting the value's
// bits from offset zero. Only correct when the two layouts agree. When an ABI flattens an aggregate
// it drops padding, and the dense type it produces puts the surviving members at different offsets
// than they really have, eg `struct #min_field_align(16){i8, f32}` flattens to `{i8, float}`, moving
// the float from offset 16 to offset 4. These give the real byte offset of each `cast_type` element.
i64 *coerce_offsets;
isize coerce_offset_count;
};
@@ -25,6 +33,14 @@ gb_internal i64 lb_alignof(LLVMTypeRef type);
gb_internal lbArgType lb_arg_type_direct(LLVMTypeRef type, LLVMTypeRef cast_type, LLVMTypeRef pad_type, LLVMAttributeRef attr) {
return lbArgType{lbArg_Direct, type, cast_type, pad_type, attr, nullptr, 0, false};
}
// Same as above, except coercion reads each element of `cast_type` from its real offset in `type`
// instead of reinterpreting the bits from offset zero. See `coerce_offsets`.
gb_internal lbArgType lb_arg_type_direct_fields(LLVMTypeRef type, LLVMTypeRef cast_type, i64 *offsets, isize count) {
lbArgType arg = lb_arg_type_direct(type, cast_type, nullptr, nullptr);
arg.coerce_offsets = offsets;
arg.coerce_offset_count = count;
return arg;
}
gb_internal lbArgType lb_arg_type_direct(LLVMTypeRef type) {
return lb_arg_type_direct(type, nullptr, nullptr, nullptr);
}
@@ -2030,6 +2046,78 @@ namespace lbAbiRiscv64 {
}
}
// `flatten` records which members survive; this records where they are. The two are walked
// together so the dense type it builds can still be read from the real object: an over-aligned
// member leaves a gap that the flatten removes, and reinterpreting the bits from offset zero
// then reads the member from where the padding used to be.
gb_internal void flatten_offsets(lbModule *m, Array<i64> *offsets, LLVMTypeRef type, i64 base) {
switch (LLVMGetTypeKind(type)) {
case LLVMStructTypeKind: {
if (LLVMIsPackedStruct(type)) {
array_add(offsets, base);
break;
}
unsigned elem_count = LLVMCountStructElementTypes(type);
// element offsets the way `lb_alignof` models LLVM's own layout
auto elem_offsets = array_make<i64>(temporary_allocator(), 0, elem_count);
i64 off = 0;
for (unsigned i = 0; i < elem_count; i += 1) {
LLVMTypeRef et = LLVMStructGetTypeAtIndex(type, i);
i64 a = lb_alignof(et);
if (a > 0) {
off = align_formula(off, a);
}
array_add(&elem_offsets, off);
off += lb_sizeof(et);
}
auto field_remapping = map_get(&m->struct_field_remapping, cast(void *)type);
if (field_remapping) {
auto remap = *field_remapping;
for_array(i, remap) {
flatten_offsets(m, offsets, LLVMStructGetTypeAtIndex(type, remap[i]), base + elem_offsets[remap[i]]);
}
break;
}
for (unsigned i = 0; i < elem_count; i += 1) {
flatten_offsets(m, offsets, LLVMStructGetTypeAtIndex(type, i), base + elem_offsets[i]);
}
break;
}
case LLVMArrayTypeKind: {
unsigned len = LLVMGetArrayLength(type);
LLVMTypeRef elem = OdinLLVMGetArrayElementType(type);
i64 stride = lb_sizeof(elem);
for (unsigned i = 0; i < len; i += 1) {
flatten_offsets(m, offsets, elem, base + cast(i64)i*stride);
}
break;
}
default:
array_add(offsets, base);
}
}
// The offsets the dense `cast_type` implies, so the two can be compared. When they agree the
// ordinary bit-reinterpreting coercion is right and nothing needs to change.
gb_internal bool flatten_moved_a_member(Array<LLVMTypeRef> const &fields, Array<i64> const &offsets) {
if (fields.count != offsets.count) {
return false;
}
i64 off = 0;
for_array(i, fields) {
i64 a = lb_alignof(fields[i]);
if (a > 0) {
off = align_formula(off, a);
}
if (off != offsets[i]) {
return true;
}
off += lb_sizeof(fields[i]);
}
return false;
}
// The psABI's rule is "one floating-point real and one integer (or bitfield)", and a pointer
// is not an integer. `is_register` admits pointers and keeps that meaning for its other
// callers, so the floating-point arms need their own predicate.
@@ -2070,10 +2158,22 @@ namespace lbAbiRiscv64 {
LLVMTypeRef fp_type = type;
LLVMTypeKind fp_kind = kind;
i64 fp_size = size;
i64 *fp_offsets = nullptr;
isize fp_offset_count = 0;
if (kind == LLVMStructTypeKind) {
Array<LLVMTypeRef> fields = array_make<LLVMTypeRef>(temporary_allocator(), 0, LLVMCountStructElementTypes(type));
flatten(m, &fields, type, false);
auto offsets = array_make<i64>(temporary_allocator(), 0, fields.count);
flatten_offsets(m, &offsets, type, 0);
if (flatten_moved_a_member(fields, offsets)) {
fp_offsets = gb_alloc_array(permanent_allocator(), i64, offsets.count);
for_array(i, offsets) {
fp_offsets[i] = offsets[i];
}
fp_offset_count = offsets.count;
}
if (fields.count == 1) {
fp_type = fields[0];
} else {
@@ -2089,6 +2189,9 @@ namespace lbAbiRiscv64 {
if (fp_type != orig_type) {
// A struct that flattened to a single float has to be coerced to that float;
// handing back the original sends an over-aligned one to integer registers.
if (fp_offset_count > 0) {
return lb_arg_type_direct_fields(orig_type, fp_type, fp_offsets, fp_offset_count);
}
return lb_arg_type_direct(orig_type, fp_type, nullptr, nullptr);
}
return non_struct(c, orig_type);
@@ -2104,18 +2207,27 @@ namespace lbAbiRiscv64 {
if (is_float(ty1) && is_float(ty2) && ty1s <= flen && ty2s <= flen && *fprs_left >= 2) {
*fprs_left -= 2;
if (fp_offset_count > 0) {
return lb_arg_type_direct_fields(orig_type, fp_type, fp_offsets, fp_offset_count);
}
return lb_arg_type_direct(orig_type, fp_type, nullptr, nullptr);
}
if (is_float(ty1) && is_int_member(ty2) && ty1s <= flen && ty2s <= xlen && *fprs_left >= 1 && *gprs_left >= 1) {
*fprs_left -= 1;
*gprs_left -= 1;
if (fp_offset_count > 0) {
return lb_arg_type_direct_fields(orig_type, fp_type, fp_offsets, fp_offset_count);
}
return lb_arg_type_direct(orig_type, fp_type, nullptr, nullptr);
}
if (is_int_member(ty1) && is_float(ty2) && ty1s <= xlen && ty2s <= flen && *gprs_left >= 1 && *fprs_left >= 1) {
*fprs_left -= 1;
*gprs_left -= 1;
if (fp_offset_count > 0) {
return lb_arg_type_direct_fields(orig_type, fp_type, fp_offsets, fp_offset_count);
}
return lb_arg_type_direct(orig_type, fp_type, nullptr, nullptr);
}
}

View File

@@ -1,3 +1,6 @@
gb_internal LLVMValueRef lb_coerce_fields_load(lbProcedure *p, lbValue x, lbArgType const *arg);
gb_internal LLVMValueRef lb_coerce_fields_store(lbProcedure *p, LLVMValueRef coerced, Type *original_type, lbArgType const *arg);
gb_internal LLVMValueRef lb_call_intrinsic(lbProcedure *p, const char *name, LLVMValueRef* args, unsigned arg_count, LLVMTypeRef* types, unsigned type_count) {
unsigned id = LLVMLookupIntrinsicID(name, gb_strlen(name));
GB_ASSERT_MSG(id != 0, "Unable to find %s", name);
@@ -681,7 +684,9 @@ gb_internal void lb_begin_procedure_body(lbProcedure *p) {
if (e->token.string.len != 0 && !is_blank_ident(e->token.string)) {
LLVMTypeRef param_type = lb_type(p->module, e->type);
LLVMValueRef original_value = LLVMGetParam(p->value, param_offset+llvm_param_index);
LLVMValueRef value = OdinLLVMBuildTransmute(p, original_value, param_type);
LLVMValueRef value = arg_type->coerce_offset_count > 0
? lb_coerce_fields_store(p, original_value, e->type, arg_type)
: OdinLLVMBuildTransmute(p, original_value, param_type);
lbValue param = {};
param.value = value;
@@ -922,6 +927,60 @@ gb_internal Array<lbValue> lb_value_to_array(lbProcedure *p, gbAllocator const &
// A `cast_type` is normally applied by reinterpreting the value's bits from offset zero. When the
// ABI flattened an aggregate and dropped padding, the dense type it produced puts the survivors at
// different offsets than they really have, and these two read and write them where they actually
// live. See `lbArgType::coerce_offsets`.
gb_internal LLVMValueRef lb_coerce_fields_load(lbProcedure *p, lbValue x, lbArgType const *arg) {
LLVMContextRef ctx = p->module->ctx;
LLVMTypeRef i8 = LLVMInt8TypeInContext(ctx);
LLVMTypeRef i64t = LLVMInt64TypeInContext(ctx);
lbValue base = lb_address_from_load_or_generate_local(p, x);
if (LLVMGetTypeKind(arg->cast_type) != LLVMStructTypeKind) {
GB_ASSERT(arg->coerce_offset_count == 1);
LLVMValueRef index = LLVMConstInt(i64t, cast(unsigned long long)arg->coerce_offsets[0], false);
LLVMValueRef ptr = LLVMBuildInBoundsGEP2(p->builder, i8, base.value, &index, 1, "");
return LLVMBuildLoad2(p->builder, arg->cast_type, ptr, "");
}
unsigned count = LLVMCountStructElementTypes(arg->cast_type);
GB_ASSERT(cast(isize)count == arg->coerce_offset_count);
LLVMValueRef result = LLVMGetUndef(arg->cast_type);
for (unsigned i = 0; i < count; i += 1) {
LLVMTypeRef elem_type = LLVMStructGetTypeAtIndex(arg->cast_type, i);
LLVMValueRef index = LLVMConstInt(i64t, cast(unsigned long long)arg->coerce_offsets[i], false);
LLVMValueRef ptr = LLVMBuildInBoundsGEP2(p->builder, i8, base.value, &index, 1, "");
LLVMValueRef elem = LLVMBuildLoad2(p->builder, elem_type, ptr, "");
result = LLVMBuildInsertValue(p->builder, result, elem, i, "");
}
return result;
}
// The reverse: scatter a coerced value back into a full-sized object of the original type.
gb_internal LLVMValueRef lb_coerce_fields_store(lbProcedure *p, LLVMValueRef coerced, Type *original_type, lbArgType const *arg) {
LLVMContextRef ctx = p->module->ctx;
LLVMTypeRef i8 = LLVMInt8TypeInContext(ctx);
LLVMTypeRef i64t = LLVMInt64TypeInContext(ctx);
lbAddr slot = lb_add_local_generated(p, original_type, true);
unsigned count = 1;
bool is_struct = LLVMGetTypeKind(arg->cast_type) == LLVMStructTypeKind;
if (is_struct) {
count = LLVMCountStructElementTypes(arg->cast_type);
}
GB_ASSERT(cast(isize)count == arg->coerce_offset_count);
for (unsigned i = 0; i < count; i += 1) {
LLVMValueRef elem = is_struct ? LLVMBuildExtractValue(p->builder, coerced, i, "") : coerced;
LLVMValueRef index = LLVMConstInt(i64t, cast(unsigned long long)arg->coerce_offsets[i], false);
LLVMValueRef ptr = LLVMBuildInBoundsGEP2(p->builder, i8, slot.addr.value, &index, 1, "");
LLVMBuildStore(p->builder, elem, ptr);
}
return lb_addr_load(p, slot).value;
}
gb_internal lbValue lb_emit_call_internal(lbProcedure *p, lbValue value, lbValue return_ptr, Array<lbValue> const &processed_args, Type *abi_rt, lbAddr context_ptr, ProcInlining inlining, ProcTailing tailing) {
GB_ASSERT(p->module->ctx == LLVMGetTypeContext(LLVMTypeOf(value.value)));
@@ -1191,7 +1250,10 @@ gb_internal lbValue lb_emit_call(lbProcedure *p, lbValue value, Array<lbValue> c
if (!abi_type) {
abi_type = arg->type;
}
if (xt == abi_type) {
if (arg->coerce_offset_count > 0) {
x.value = lb_coerce_fields_load(p, x, arg);
array_add(&processed_args, x);
} else if (xt == abi_type) {
array_add(&processed_args, x);
} else {
x.value = OdinLLVMBuildTransmute(p, x.value, abi_type);
@@ -1261,10 +1323,14 @@ gb_internal lbValue lb_emit_call(lbProcedure *p, lbValue value, Array<lbValue> c
result = lb_emit_load(p, return_ptr);
} else if (rt != nullptr) {
result = lb_emit_call_internal(p, value, {}, processed_args, rt, context_ptr, inlining, tailing);
if (ft->ret.cast_type) {
result.value = OdinLLVMBuildTransmute(p, result.value, ft->ret.cast_type);
if (ft->ret.coerce_offset_count > 0) {
result.value = lb_coerce_fields_store(p, result.value, rt, &ft->ret);
} else {
if (ft->ret.cast_type) {
result.value = OdinLLVMBuildTransmute(p, result.value, ft->ret.cast_type);
}
result.value = OdinLLVMBuildTransmute(p, result.value, ft->ret.type);
}
result.value = OdinLLVMBuildTransmute(p, result.value, ft->ret.type);
result.type = rt;
if (LLVMTypeOf(result.value) == LLVMInt1TypeInContext(p->module->ctx)) {
result.type = t_llvm_bool;