From e5cc35c44554caa4f9f56a7772269bbe48c3eabb Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 11:02:55 +0100 Subject: [PATCH] Lower what LLVM's fast instruction selector cannot select before emitting objects, and declare debug variables through stack slots --- src/llvm_backend.cpp | 529 ++++++++++++++++++ src/llvm_backend_debug.cpp | 21 + src/llvm_backend_general.cpp | 55 +- src/llvm_backend_proc.cpp | 7 +- src/llvm_backend_utility.cpp | 21 + tests/issues/run.bat | 3 + tests/issues/run.sh | 3 + .../issues/test_issue_fast_isel_lowering.odin | 121 ++++ 8 files changed, 750 insertions(+), 10 deletions(-) create mode 100644 tests/issues/test_issue_fast_isel_lowering.odin diff --git a/src/llvm_backend.cpp b/src/llvm_backend.cpp index d39318215..6acc3c723 100644 --- a/src/llvm_backend.cpp +++ b/src/llvm_backend.cpp @@ -2545,6 +2545,531 @@ gb_internal bool lb_is_module_empty(lbModule *m) { return true; } +// NOTE(bill, 2026-10-03) +// +// LLVM's fast instruction selector cannot select a first class aggregate `load`, `store`, `insertvalue`, or `select`, +// and hands the rest of the block to SelectionDAG. +// SROA leaves many behind so it stores/loads the fields instead +enum { + LB_SCALARIZE_MAX_LEAVES = 32, + LB_SCALARIZE_MAX_DEPTH = 8, +}; + +struct lbAggregateLeaf { + unsigned path[LB_SCALARIZE_MAX_DEPTH]; + unsigned depth; +}; + +struct lbScalarizedPhi { + LLVMValueRef aggregate; + unsigned path[LB_SCALARIZE_MAX_DEPTH]; + unsigned depth; + LLVMValueRef field; + bool ok; +}; + +struct lbFastIselLowering { + lbModule * m; + LLVMBuilderRef builder; + LLVMValueRef store; + Array phis; +}; + +gb_internal i64 lb_aggregate_path_offset(LLVMTypeRef type, unsigned const *path, unsigned depth, LLVMTypeRef *leaf_type_) { + i64 offset = 0; + for (unsigned d = 0; d < depth; d++) { + if (LLVMGetTypeKind(type) == LLVMStructTypeKind) { + bool is_packed = LLVMIsPackedStruct(type); + i64 field_offset = 0; + for (unsigned i = 0; i <= path[d]; i++) { + LLVMTypeRef field = LLVMStructGetTypeAtIndex(type, i); + if (!is_packed) { + field_offset = llvm_align_formula(field_offset, lb_alignof(field)); + } + if (i == path[d]) { + type = field; + break; + } + field_offset += lb_sizeof(field); + } + offset += field_offset; + } else { + type = OdinLLVMGetArrayElementType(type); + offset += cast(i64)path[d] * lb_sizeof(type); + } + } + if (leaf_type_) *leaf_type_ = type; + return offset; +} + +gb_internal bool lb_aggregate_leaves(LLVMTypeRef type, unsigned *path, unsigned depth, lbAggregateLeaf *leaves, isize *leaf_count) { + LLVMTypeKind kind = LLVMGetTypeKind(type); + if (kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind) { + if (depth >= LB_SCALARIZE_MAX_DEPTH) { + return false; + } + if (kind == LLVMStructTypeKind && LLVMIsOpaqueStruct(type)) { + return false; + } + unsigned count = kind == LLVMStructTypeKind ? LLVMCountStructElementTypes(type) : cast(unsigned)LLVMGetArrayLength(type); + for (unsigned i = 0; i < count; i++) { + path[depth] = i; + LLVMTypeRef elem = kind == LLVMStructTypeKind ? LLVMStructGetTypeAtIndex(type, i) : OdinLLVMGetArrayElementType(type); + if (!lb_aggregate_leaves(elem, path, depth+1, leaves, leaf_count)) { + return false; + } + } + return true; + } + + if (*leaf_count >= LB_SCALARIZE_MAX_LEAVES) { + return false; + } + + lbAggregateLeaf *leaf = &leaves[(*leaf_count)++]; + gb_memmove(leaf->path, path, depth*gb_size_of(unsigned)); + leaf->depth = depth; + return true; +} + +gb_internal unsigned lb_alignment_at_offset(unsigned alignment, i64 offset) { + unsigned a = gb_max(alignment, 1u); + while (offset % a != 0) { + a >>= 1; + } + return a; +} + +gb_internal LLVMValueRef lb_aggregate_field_gep(lbModule *m, LLVMBuilderRef b, LLVMTypeRef type, LLVMValueRef ptr, unsigned const *path, unsigned depth) { + LLVMValueRef indices[LB_SCALARIZE_MAX_DEPTH+1] = {}; + LLVMTypeRef i32 = LLVMInt32TypeInContext(m->ctx); + indices[0] = LLVMConstInt(i32, 0, false); + for (unsigned d = 0; d < depth; d++) { + indices[d+1] = LLVMConstInt(i32, path[d], false); + } + return LLVMBuildInBoundsGEP2(b, type, ptr, indices, depth+1, ""); +} + +// returns false if the field cannot be reached without an aggregate value, and sets `*field_` to nullptr when it is undefined +gb_internal bool lb_aggregate_field_value(lbFastIselLowering *s, LLVMValueRef v, unsigned const *path, unsigned depth, LLVMValueRef *field_) { + if (depth == 0) { + *field_ = (LLVMIsUndef(v) || LLVMIsPoison(v)) ? nullptr : v; + return true; + } + if (LLVMIsUndef(v) || LLVMIsPoison(v)) { + *field_ = nullptr; + return true; + } + + if (LLVMIsAConstant(v)) { + LLVMValueRef elem = LLVMGetAggregateElement(v, path[0]); + if (elem != nullptr) { + return lb_aggregate_field_value(s, elem, path+1, depth-1, field_); + } + } else if (LLVMIsAInsertValueInst(v)) { + unsigned n = LLVMGetNumIndices(v); + unsigned const *indices = LLVMGetIndices(v); + + unsigned k = 0; + while (k < n && k < depth && indices[k] == path[k]) { + k += 1; + } + + if (k == n) { + return lb_aggregate_field_value(s, LLVMGetOperand(v, 1), path+n, depth-n, field_); + } else if (k < depth) { + return lb_aggregate_field_value(s, LLVMGetOperand(v, 0), path, depth, field_); + } + return false; + } else if (LLVMIsAExtractValueInst(v)) { + unsigned n = LLVMGetNumIndices(v); + if (n + depth <= LB_SCALARIZE_MAX_DEPTH) { + unsigned full[LB_SCALARIZE_MAX_DEPTH]; + gb_memmove(full, LLVMGetIndices(v), n*gb_size_of(unsigned)); + gb_memmove(full+n, path, depth*gb_size_of(unsigned)); + return lb_aggregate_field_value(s, LLVMGetOperand(v, 0), full, n+depth, field_); + } + return false; + } else if (LLVMIsASelectInst(v)) { + LLVMValueRef x = nullptr; + LLVMValueRef y = nullptr; + if (!lb_aggregate_field_value(s, LLVMGetOperand(v, 1), path, depth, &x) || + !lb_aggregate_field_value(s, LLVMGetOperand(v, 2), path, depth, &y)) { + return false; + } + if (x == nullptr || y == nullptr) { + *field_ = x ? x : y; + return true; + } + LLVMPositionBuilderBefore(s->builder, s->store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(s->store)); + *field_ = LLVMBuildSelect(s->builder, LLVMGetOperand(v, 0), x, y, ""); + return true; + } else if (LLVMIsAPHINode(v)) { + for (lbScalarizedPhi const &e : s->phis) { + if (e.aggregate == v && + e.depth == depth && + gb_memcompare(e.path, path, depth*gb_size_of(unsigned)) == 0) { + *field_ = e.field; + return e.ok; + } + } + + LLVMTypeRef field_type = nullptr; + lb_aggregate_path_offset(LLVMTypeOf(v), path, depth, &field_type); + + LLVMPositionBuilderBefore(s->builder, v); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + + // NOTE(bill): cached before its incoming values, as a loop reaches it again + isize index = s->phis.count; + lbScalarizedPhi entry = {}; + entry.aggregate = v; + gb_memmove(entry.path, path, depth*gb_size_of(unsigned)); + + LLVMValueRef phi = LLVMBuildPhi(s->builder, field_type, ""); + entry.depth = depth; + entry.field = phi; + entry.ok = true; + array_add(&s->phis, entry); + + LLVMValueRef saved_store = s->store; + bool ok = true; + unsigned incoming_count = LLVMCountIncoming(v); + for (unsigned k = 0; k < incoming_count; k++) { + LLVMBasicBlockRef block = LLVMGetIncomingBlock(v, k); + s->store = LLVMGetBasicBlockTerminator(block); + LLVMValueRef field = nullptr; + if (!lb_aggregate_field_value(s, LLVMGetIncomingValue(v, k), path, depth, &field)) { + ok = false; + field = nullptr; + } + if (field == nullptr) { + field = LLVMGetUndef(field_type); + } + LLVMAddIncoming(phi, &field, &block, 1); + } + s->store = saved_store; + s->phis[index].ok = ok; + *field_ = phi; + return ok; + } else if (LLVMIsALoadInst(v) && !LLVMGetVolatile(v) && LLVMGetOrdering(v) == LLVMAtomicOrderingNotAtomic) { + // NOTE(bill): Read the field where the aggregate was read, as the memory may change before the store + LLVMTypeRef type = LLVMTypeOf(v); + LLVMTypeRef field_type = nullptr; + i64 offset = lb_aggregate_path_offset(type, path, depth, &field_type); + + LLVMPositionBuilderBefore(s->builder, LLVMGetNextInstruction(v)); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(v)); + + LLVMValueRef ptr = lb_aggregate_field_gep(s->m, s->builder, type, LLVMGetOperand(v, 0), path, depth); + LLVMValueRef field = LLVMBuildLoad2(s->builder, field_type, ptr, ""); + LLVMSetAlignment(field, lb_alignment_at_offset(LLVMGetAlignment(v), offset)); + + *field_ = field; + return true; + } + + if (depth == 1) { + LLVMPositionBuilderBefore(s->builder, s->store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(s->store)); + *field_ = LLVMBuildExtractValue(s->builder, v, path[0], ""); + return true; + } + return false; +} + +gb_internal bool lb_scalarize_aggregate_store(lbFastIselLowering *s, LLVMValueRef store) { + LLVMValueRef value = LLVMGetOperand(store, 0); + LLVMValueRef ptr = LLVMGetOperand(store, 1); + LLVMTypeRef type = LLVMTypeOf(value); + + lbAggregateLeaf leaves[LB_SCALARIZE_MAX_LEAVES] = {}; + isize leaf_count = 0; + unsigned path[LB_SCALARIZE_MAX_DEPTH] = {}; + + if (!lb_aggregate_leaves(type, path, 0, leaves, &leaf_count)) { + return false; + } + s->store = store; + + LLVMValueRef fields[LB_SCALARIZE_MAX_LEAVES] = {}; + for (isize i = 0; i < leaf_count; i++) { + if (!lb_aggregate_field_value(s, value, leaves[i].path, leaves[i].depth, &fields[i])) { + return false; + } + } + + unsigned alignment = LLVMGetAlignment(store); + + LLVMPositionBuilderBefore(s->builder, store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); + + for (isize i = 0; i < leaf_count; i++) { + if (fields[i] == nullptr) { + continue; + } + i64 offset = lb_aggregate_path_offset(type, leaves[i].path, leaves[i].depth, nullptr); + LLVMValueRef field_ptr = lb_aggregate_field_gep(s->m, s->builder, type, ptr, leaves[i].path, leaves[i].depth); + LLVMValueRef field_store = LLVMBuildStore(s->builder, fields[i], field_ptr); + LLVMSetAlignment(field_store, lb_alignment_at_offset(alignment, offset)); + } + LLVMInstructionEraseFromParent(store); + return true; +} + +gb_internal bool lb_is_aggregate_type(LLVMTypeRef type) { + LLVMTypeKind kind = LLVMGetTypeKind(type); + return kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind; +} + +gb_internal bool lb_is_plain_access(LLVMValueRef inst) { + return !LLVMGetVolatile(inst) && LLVMGetOrdering(inst) == LLVMAtomicOrderingNotAtomic; +} + +gb_internal void lb_lower_bool_select(lbFastIselLowering *s, LLVMValueRef select) { + LLVMValueRef c = LLVMGetOperand(select, 0); + LLVMValueRef x = LLVMGetOperand(select, 1); + LLVMValueRef y = LLVMGetOperand(select, 2); + if (LLVMGetTypeKind(LLVMTypeOf(c)) != LLVMIntegerTypeKind) { + return; + } + LLVMPositionBuilderBefore(s->builder, select); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(select)); + LLVMValueRef t = LLVMConstInt(LLVMTypeOf(c), 1, false); + + LLVMValueRef res = nullptr; + if (LLVMIsAConstantInt(x)) { + res = LLVMConstIntGetZExtValue(x) ? LLVMBuildOr(s->builder, c, y, "") : LLVMBuildAnd(s->builder, LLVMBuildXor(s->builder, c, t, ""), y, ""); + } else if (LLVMIsAConstantInt(y)) { + res = LLVMConstIntGetZExtValue(y) ? LLVMBuildOr(s->builder, LLVMBuildXor(s->builder, c, t, ""), x, "") : LLVMBuildAnd(s->builder, c, x, ""); + } else { + LLVMValueRef a = LLVMBuildAnd(s->builder, c, x, ""); + LLVMValueRef b = LLVMBuildAnd(s->builder, LLVMBuildXor(s->builder, c, t, ""), y, ""); + res = LLVMBuildOr(s->builder, a, b, ""); + } + LLVMReplaceAllUsesWith(select, res); + LLVMInstructionEraseFromParent(select); +} + +gb_internal void lb_truncate_bool_arguments(lbFastIselLowering *s, LLVMValueRef call) { + LLVMTypeRef i1 = LLVMInt1TypeInContext(s->m->ctx); + LLVMTypeRef i8 = LLVMInt8TypeInContext(s->m->ctx); + LLVMBasicBlockRef block = LLVMGetInstructionParent(call); + unsigned arg_count = LLVMGetNumArgOperands(call); + for (unsigned a = 0; a < arg_count; a++) { + LLVMValueRef arg = LLVMGetOperand(call, a); + if (LLVMTypeOf(arg) != i1 || LLVMIsAConstantInt(arg)) { + continue; + } + LLVMUseRef first_use = LLVMGetFirstUse(arg); + if (LLVMIsATruncInst(arg) && LLVMGetInstructionParent(arg) == block && + first_use != nullptr && LLVMGetNextUse(first_use) == nullptr) { + continue; + } + LLVMPositionBuilderBefore(s->builder, call); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(call)); + LLVMValueRef byte = LLVMBuildZExt(s->builder, arg, i8, ""); + LLVMSetOperand(call, a, LLVMBuildTrunc(s->builder, byte, i1, "")); + } +} + +gb_internal void lb_lower_small_switch(lbFastIselLowering *s, LLVMValueRef sw) { + enum {MAX_CASES = 3}; + LLVMValueRef cond = LLVMGetOperand(sw, 0); + unsigned case_count = (cast(unsigned)LLVMGetNumOperands(sw) - 2) / 2; + if (case_count == 0 || case_count > MAX_CASES || LLVMGetIntTypeWidth(LLVMTypeOf(cond)) > 64) { + return; + } + + LLVMBasicBlockRef block = LLVMGetInstructionParent(sw); + LLVMBasicBlockRef next_block = LLVMGetNextBasicBlock(block); + + LLVMValueRef fn = LLVMGetBasicBlockParent(block); + LLVMMetadataRef loc = LLVMInstructionGetDebugLoc(sw); + + LLVMBasicBlockRef from[MAX_CASES+1] = {}; + LLVMBasicBlockRef to [MAX_CASES+1] = {}; + + LLVMBasicBlockRef curr = block; + + LLVMPositionBuilderBefore(s->builder, sw); + LLVMSetCurrentDebugLocation2(s->builder, loc); + + for (unsigned j = 0; j < case_count; j++) { + LLVMBasicBlockRef dest = LLVMGetSuccessor(sw, j+1); + LLVMBasicBlockRef else_block = LLVMGetSwitchDefaultDest(sw); + if (j+1 < case_count) { + if (next_block != nullptr) { + else_block = LLVMInsertBasicBlockInContext(s->m->ctx, next_block, ""); + } else { + else_block = LLVMAppendBasicBlockInContext(s->m->ctx, fn, ""); + } + } + + LLVMValueRef cmp = LLVMBuildICmp(s->builder, LLVMIntEQ, cond, LLVMGetOperand(sw, 2 + 2*j), ""); + LLVMBuildCondBr(s->builder, cmp, dest, else_block); + + from[j] = curr; + to[j] = dest; + + if (j+1 < case_count) { + curr = else_block; + LLVMPositionBuilderAtEnd(s->builder, curr); + LLVMSetCurrentDebugLocation2(s->builder, loc); + } + } + from[case_count] = curr; + to [case_count] = LLVMGetSwitchDefaultDest(sw); + LLVMInstructionEraseFromParent(sw); + + // the incoming entries for `block` become one for each new edge + for (unsigned e = 0; e <= case_count; e++) { + bool seen = false; + for (unsigned k = 0; k < e; k++) { + seen |= to[k] == to[e]; + } + if (seen) { + continue; + } + + LLVMBasicBlockRef dest = to[e]; + for (LLVMValueRef phi = LLVMGetFirstInstruction(dest); phi != nullptr && LLVMIsAPHINode(phi); /**/) { + LLVMValueRef next = LLVMGetNextInstruction(phi); + LLVMValueRef value_from_block = nullptr; + + unsigned incoming_count = LLVMCountIncoming(phi); + for (unsigned k = 0; k < incoming_count; k++) { + if (LLVMGetIncomingBlock(phi, k) == block) { + value_from_block = LLVMGetIncomingValue(phi, k); + } + } + + if (value_from_block != nullptr) { + LLVMPositionBuilderBefore(s->builder, phi); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + LLVMValueRef new_phi = LLVMBuildPhi(s->builder, LLVMTypeOf(phi), ""); + for (unsigned k = 0; k < incoming_count; k++) { + LLVMBasicBlockRef incoming_block = LLVMGetIncomingBlock(phi, k); + if (incoming_block != block) { + LLVMValueRef incoming_value = LLVMGetIncomingValue(phi, k); + LLVMAddIncoming(new_phi, &incoming_value, &incoming_block, 1); + } + } + for (unsigned k = 0; k <= case_count; k++) { + if (to[k] == dest) { + LLVMAddIncoming(new_phi, &value_from_block, &from[k], 1); + } + } + LLVMReplaceAllUsesWith(phi, new_phi); + LLVMInstructionEraseFromParent(phi); + } + + phi = next; + } + } +} + +gb_internal void lb_lower_for_fast_isel(lbModule *m) { + lbFastIselLowering s = {}; + s.m = m; + + s.builder = LLVMCreateBuilderInContext(m->ctx); + defer (LLVMDisposeBuilder(s.builder)); + + array_init(&s.phis, heap_allocator()); + defer (array_free(&s.phis)); + + auto work = array_make(heap_allocator(), 0, 64); + defer (array_free(&work)); + + for (LLVMValueRef fn = LLVMGetFirstFunction(m->mod); fn != nullptr; fn = LLVMGetNextFunction(fn)) { + array_clear(&work); + array_clear(&s.phis); + + for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { + for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; i = LLVMGetNextInstruction(i)) { + if (LLVMIsAStoreInst(i) && lb_is_plain_access(i) && lb_is_aggregate_type(LLVMTypeOf(LLVMGetOperand(i, 0)))) { + array_add(&work, i); + } + } + } + for (LLVMValueRef store : work) { + lb_scalarize_aggregate_store(&s, store); + } + + // NOTE(bill): A field read out of an aggregate value is read where that aggregate came from + array_clear(&work); + for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { + for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; i = LLVMGetNextInstruction(i)) { + if (LLVMIsAExtractValueInst(i) && !lb_is_aggregate_type(LLVMTypeOf(i))) { + array_add(&work, i); + } else if (LLVMIsASelectInst(i) && LLVMTypeOf(i) == LLVMInt1TypeInContext(m->ctx)) { + array_add(&work, i); + } else if (LLVMIsACallInst(i)) { + array_add(&work, i); + } else if (LLVMIsASwitchInst(i)) { + array_add(&work, i); + } + } + } + for (LLVMValueRef i : work) { + if (LLVMIsASelectInst(i)) { + lb_lower_bool_select(&s, i); + continue; + } + if (LLVMIsACallInst(i)) { + lb_truncate_bool_arguments(&s, i); + continue; + } + if (LLVMIsASwitchInst(i)) { + lb_lower_small_switch(&s, i); + continue; + } + LLVMValueRef agg = LLVMGetOperand(i, 0); + unsigned n = LLVMGetNumIndices(i); + if (n > LB_SCALARIZE_MAX_DEPTH) { + continue; + } + + if (!(LLVMIsAConstant(agg) || + LLVMIsAInsertValueInst(agg) || + LLVMIsASelectInst(agg) || + LLVMIsAPHINode(agg) || + (LLVMIsALoadInst(agg) && lb_is_plain_access(agg)))) { + continue; + } + + s.store = i; + + LLVMValueRef field = nullptr; + if (lb_aggregate_field_value(&s, agg, LLVMGetIndices(i), n, &field) && field != nullptr) { + LLVMReplaceAllUsesWith(i, field); + LLVMInstructionEraseFromParent(i); + } + } + + bool removed = true; + while (removed) { + removed = false; + for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { + for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; /**/) { + LLVMValueRef next = LLVMGetNextInstruction(i); + if (LLVMGetFirstUse(i) == nullptr && + (LLVMIsAInsertValueInst(i) || LLVMIsAExtractValueInst(i) || LLVMIsASelectInst(i) || LLVMIsAPHINode(i) || + LLVMIsAGetElementPtrInst(i) || (LLVMIsALoadInst(i) && lb_is_plain_access(i)))) { + LLVMInstructionEraseFromParent(i); + removed = true; + } + i = next; + } + } + } + } +} + struct lbLLVMEmitWorker { LLVMTargetMachineRef target_machine; LLVMCodeGenFileType code_gen_file_type; @@ -2743,6 +3268,10 @@ gb_internal WORKER_TASK_PROC(lb_llvm_module_pass_worker_proc) { return 1; } + if (lb_uses_fast_isel()) { + lb_lower_for_fast_isel(wd->m); + } + if (LLVM_IGNORE_VERIFICATION) { return 0; } diff --git a/src/llvm_backend_debug.cpp b/src/llvm_backend_debug.cpp index 2b1cbf872..af9f1323e 100644 --- a/src/llvm_backend_debug.cpp +++ b/src/llvm_backend_debug.cpp @@ -1206,6 +1206,25 @@ gb_internal LLVMMetadataRef lb_debug_type(lbModule *m, Type *type) { return dt; } +// A variable whose address is not a stack slot of its own (an argument, or an element or a pointer it refers to) +// would have its location tracked with `DBG_VALUE`s through every block of the procedure, +// so it is described through a stack slot holding that address instead +gb_internal LLVMValueRef lb_debug_storage(lbProcedure *p, LLVMValueRef storage, LLVMMetadataRef *expr) { + bool is_argument = LLVMIsAArgument(storage) != nullptr; + if (!is_argument && (!LLVMIsAInstruction(storage) || LLVMIsAAllocaInst(storage))) { + return storage; + } + LLVMBasicBlockRef insert_block = LLVMGetInsertBlock(p->builder); + LLVMValueRef slot = llvm_alloca(p, LLVMTypeOf(storage), build_context.ptr_size, ""); + LLVMPositionBuilderAtEnd(p->builder, is_argument ? p->decl_block->block : insert_block); + LLVMBuildStore(p->builder, storage, slot); + LLVMPositionBuilderAtEnd(p->builder, insert_block); + + uint64_t deref = 0x06; // DW_OP_deref + *expr = LLVMDIBuilderCreateExpression(p->module->debug_builder, &deref, 1); + return slot; +} + gb_internal void lb_add_debug_local_variable(lbProcedure *p, LLVMValueRef ptr, Type *type, Token const &token) { if (p->debug_info == nullptr) { return; @@ -1264,6 +1283,7 @@ gb_internal void lb_add_debug_local_variable(lbProcedure *p, LLVMValueRef ptr, T LLVMMetadataRef llvm_debug_loc = lb_debug_location_from_token_pos(p, token.pos); LLVMMetadataRef llvm_expr = LLVMDIBuilderCreateExpression(m->debug_builder, nullptr, 0); lb_set_llvm_metadata(m, ptr, llvm_expr); + storage = lb_debug_storage(p, storage, &llvm_expr); #if LLVM_VERSION_MAJOR <= 18 LLVMDIBuilderInsertDeclareAtEnd(m->debug_builder, storage, var_info, llvm_expr, llvm_debug_loc, block); @@ -1329,6 +1349,7 @@ gb_internal void lb_add_debug_param_variable(lbProcedure *p, LLVMValueRef ptr, T LLVMMetadataRef llvm_debug_loc = lb_debug_location_from_token_pos(p, token.pos); LLVMMetadataRef llvm_expr = LLVMDIBuilderCreateExpression(m->debug_builder, nullptr, 0); lb_set_llvm_metadata(m, ptr, llvm_expr); + storage = lb_debug_storage(p, storage, &llvm_expr); // NOTE(bill, 2022-02-01): For parameter values, you must insert them at the end of the decl block // The reason is that if the parameter is at index 0 and a pointer, there is not such things as an diff --git a/src/llvm_backend_general.cpp b/src/llvm_backend_general.cpp index 417504376..16fee3c8f 100644 --- a/src/llvm_backend_general.cpp +++ b/src/llvm_backend_general.cpp @@ -1814,6 +1814,43 @@ gb_internal bool lb_is_type_proc_recursive(Type *t) { } } +// LLVM's fast instruction selector, used for unoptimized code, hands anything it cannot select +// (such as `llvm.memmove` or the `*.inline` intrinsics) to SelectionDAG, which is far slower +gb_internal bool lb_uses_fast_isel(void) { + return (build_context.optimization_level <= 0 || build_context.fast_isel) && is_arch_x86(); +} + +gb_internal void lb_emit_memmove(lbProcedure *p, LLVMValueRef dst, unsigned dst_align, LLVMValueRef src, unsigned src_align, LLVMValueRef len) { + if (!lb_uses_fast_isel()) { + LLVMBuildMemMove(p->builder, dst, dst_align, src, src_align, len); + return; + } + lbModule *m = p->module; + LLVMTypeRef ptr_type = lb_type(m, t_rawptr); + LLVMTypeRef int_type = lb_type(m, t_int); + + if (LLVMIsConstant(len)) { + i64 size = cast(i64)LLVMConstIntGetZExtValue(len); + if (size == 1 || size == 2 || size == 4 || size == 8) { + LLVMTypeRef chunk_type = LLVMIntTypeInContext(m->ctx, cast(unsigned)(8*size)); + LLVMValueRef value = LLVMBuildLoad2(p->builder, chunk_type, src, ""); + LLVMSetAlignment(value, gb_max(src_align, 1u)); + LLVMValueRef store = LLVMBuildStore(p->builder, value, dst); + LLVMSetAlignment(store, gb_max(dst_align, 1u)); + return; + } + } + + LLVMTypeRef params[3] = {ptr_type, ptr_type, int_type}; + LLVMTypeRef proc_type = LLVMFunctionType(ptr_type, params, gb_count_of(params), false); + LLVMValueRef proc = LLVMGetNamedFunction(m->mod, "memmove"); + if (proc == nullptr) { + proc = LLVMAddFunction(m->mod, "memmove", proc_type); + } + LLVMValueRef args[3] = {dst, src, LLVMBuildIntCast2(p->builder, len, int_type, false, "")}; + LLVMBuildCall2(p->builder, proc_type, proc, args, gb_count_of(args), ""); +} + // LLVM's fast instruction selector cannot select an aggregate load or store // Oh, how do I love LLVM /s gb_internal bool lb_copies_aggregates_as_scalars(lbProcedure *p) { @@ -1863,7 +1900,7 @@ gb_internal bool lb_try_copy_loaded_aggregate(lbProcedure *p, LLVMValueRef dst, LLVMTypeRef i64_type = LLVMInt64TypeInContext(ctx); if (size > MAX_SCALAR_COPY_SIZE) { - LLVMBuildMemMove(p->builder, dst, lb_try_get_alignment(dst, 1), src, LLVMGetAlignment(load), LLVMConstInt(i64_type, size, false)); + lb_emit_memmove(p, dst, lb_try_get_alignment(dst, 1), src, LLVMGetAlignment(load), LLVMConstInt(i64_type, size, false)); return true; } @@ -1941,10 +1978,10 @@ gb_internal void lb_emit_store(lbProcedure *p, lbValue ptr, lbValue value) { LLVMValueRef src_ptr_original = LLVMGetOperand(value.value, 0); LLVMValueRef src_ptr = LLVMBuildPointerCast(p->builder, src_ptr_original, LLVMTypeOf(dst_ptr), ""); - LLVMBuildMemMove(p->builder, - dst_ptr, lb_try_get_alignment(dst_ptr, 1), - src_ptr, lb_try_get_alignment(src_ptr_original, 1), - LLVMConstInt(LLVMInt64TypeInContext(p->module->ctx), lb_sizeof(LLVMTypeOf(value.value)), false)); + lb_emit_memmove(p, + dst_ptr, lb_try_get_alignment(dst_ptr, 1), + src_ptr, lb_try_get_alignment(src_ptr_original, 1), + LLVMConstInt(LLVMInt64TypeInContext(p->module->ctx), lb_sizeof(LLVMTypeOf(value.value)), false)); return; } else if (LLVMIsConstant(value.value)) { lbAddr addr = lb_add_global_generated_from_procedure(p, value.type, value); @@ -1954,10 +1991,10 @@ gb_internal void lb_emit_store(lbProcedure *p, lbValue ptr, lbValue value) { LLVMValueRef src_ptr = addr.addr.value; src_ptr = LLVMBuildPointerCast(p->builder, src_ptr, LLVMTypeOf(dst_ptr), ""); - LLVMBuildMemMove(p->builder, - dst_ptr, lb_try_get_alignment(dst_ptr, 1), - src_ptr, lb_try_get_alignment(src_ptr, 1), - LLVMConstInt(LLVMInt64TypeInContext(p->module->ctx), lb_sizeof(LLVMTypeOf(value.value)), false)); + LLVMBuildMemCpy(p->builder, + dst_ptr, lb_try_get_alignment(dst_ptr, 1), + src_ptr, lb_try_get_alignment(src_ptr, 1), + LLVMConstInt(LLVMInt64TypeInContext(p->module->ctx), lb_sizeof(LLVMTypeOf(value.value)), false)); return; } } diff --git a/src/llvm_backend_proc.cpp b/src/llvm_backend_proc.cpp index a5f8ba66a..3172bc0b0 100644 --- a/src/llvm_backend_proc.cpp +++ b/src/llvm_backend_proc.cpp @@ -17,6 +17,11 @@ gb_internal void lb_mem_copy_overlapping(lbProcedure *p, lbValue dst, lbValue sr src = lb_emit_conv(p, src, t_rawptr); len = lb_emit_conv(p, len, t_int); + if (!is_volatile && lb_uses_fast_isel()) { + lb_emit_memmove(p, dst.value, 1, src.value, 1, len.value); + return; + } + char const *name = "llvm.memmove"; if (!p->is_startup && LLVMIsConstant(len.value)) { i64 const_len = cast(i64)LLVMConstIntGetSExtValue(len.value); @@ -47,7 +52,7 @@ gb_internal void lb_mem_copy_non_overlapping(lbProcedure *p, lbValue dst, lbValu len = lb_emit_conv(p, len, t_int); char const *name = "llvm.memcpy"; - if (!p->is_startup && LLVMIsConstant(len.value)) { + if (!p->is_startup && !lb_uses_fast_isel() && LLVMIsConstant(len.value)) { i64 const_len = cast(i64)LLVMConstIntGetSExtValue(len.value); if (const_len <= lb_max_zero_init_size()) { name = "llvm.memcpy.inline"; diff --git a/src/llvm_backend_utility.cpp b/src/llvm_backend_utility.cpp index 19b203e3c..aa33073c5 100644 --- a/src/llvm_backend_utility.cpp +++ b/src/llvm_backend_utility.cpp @@ -113,6 +113,27 @@ gb_internal LLVMValueRef lb_mem_zero_ptr_internal(lbProcedure *p, LLVMValueRef p } + if (is_inlinable && !is_volatile && lb_uses_fast_isel()) { + LLVMValueRef dst = LLVMBuildPointerCast(p->builder, ptr, lb_type(p->module, t_rawptr), ""); + LLVMValueRef last = nullptr; + for (i64 offset = 0; offset < const_len; /**/) { + i64 chunk = 8; + while (chunk > const_len - offset) { + chunk >>= 1; + } + LLVMTypeRef chunk_type = LLVMIntTypeInContext(p->module->ctx, cast(unsigned)(8*chunk)); + LLVMValueRef chunk_ptr = dst; + if (offset != 0) { + LLVMValueRef index = LLVMConstInt(lb_type(p->module, t_int), offset, false); + chunk_ptr = LLVMBuildGEP2(p->builder, LLVMInt8TypeInContext(p->module->ctx), dst, &index, 1, ""); + } + last = LLVMBuildStore(p->builder, LLVMConstNull(chunk_type), chunk_ptr); + LLVMSetAlignment(last, 1); + offset += chunk; + } + return last; + } + char const *name = "llvm.memset"; if (is_inlinable) { name = "llvm.memset.inline"; diff --git a/tests/issues/run.bat b/tests/issues/run.bat index f7538ca21..939041d70 100644 --- a/tests/issues/run.bat +++ b/tests/issues/run.bat @@ -95,6 +95,9 @@ clang -c ..\test_issue_sysv_abi.c -o test_issue_sysv_abi_c.o || exit /b ..\..\..\odin run ..\test_issue_7596.odin %COMMON% || exit /b ..\..\..\odin test ..\test_issue_split_globals -define:ODIN_TEST_FANCY=false -vet -strict-style -ignore-unused-defineables || exit /b ..\..\..\odin test ..\test_issue_split_globals -define:ODIN_TEST_FANCY=false -vet -strict-style -ignore-unused-defineables -debug || exit /b +..\..\..\odin test ..\test_issue_omitted_field_union.odin %COMMON% || exit /b +..\..\..\odin test ..\test_issue_fast_isel_lowering.odin %COMMON% || exit /b +..\..\..\odin test ..\test_issue_fast_isel_lowering.odin %COMMON% -o:none || exit /b @echo off diff --git a/tests/issues/run.sh b/tests/issues/run.sh index 523c52322..0207e012e 100755 --- a/tests/issues/run.sh +++ b/tests/issues/run.sh @@ -121,6 +121,9 @@ $ODIN test ../test_issue_7587.odin $COMMON $ODIN run ../test_issue_7596.odin $COMMON $ODIN test ../test_issue_split_globals -define:ODIN_TEST_FANCY=false -vet -strict-style -ignore-unused-defineables $ODIN test ../test_issue_split_globals -define:ODIN_TEST_FANCY=false -vet -strict-style -ignore-unused-defineables -debug +$ODIN test ../test_issue_omitted_field_union.odin $COMMON +$ODIN test ../test_issue_fast_isel_lowering.odin $COMMON +$ODIN test ../test_issue_fast_isel_lowering.odin $COMMON -o:none $ODIN test ../test_issue_7421.odin $COMMON if [[ $($ODIN check ../test_issue_7421_tagged_duplicate.odin $COMMON_CHECK 2>&1 >/dev/null | grep -c "Error: Duplicate case") -eq 1 ]]; then echo "SUCCESSFUL 1/1" diff --git a/tests/issues/test_issue_fast_isel_lowering.odin b/tests/issues/test_issue_fast_isel_lowering.odin new file mode 100644 index 000000000..138371a15 --- /dev/null +++ b/tests/issues/test_issue_fast_isel_lowering.odin @@ -0,0 +1,121 @@ +// What unoptimized x86 code is lowered to before LLVM's fast instruction selector sees it +package test_issues + +import "core:testing" + +Pair :: struct { + name: string, + value: int, +} + +Wide :: struct { + a, b, c: i32, + d: [3]f32, + e: Pair, + f: bool, +} + +@(require_results) +pick :: proc(cond: bool, x, y: Pair) -> Pair { + return x if cond else y +} + +@(require_results) +copy_through :: proc(dst, src: ^Wide) -> Wide { + dst^ = src^ + return dst^ +} + +@(require_results) +classify :: proc(x: int) -> (res: string) { + switch x { + case 1: res = "one" + case 2: res = "two" + case 7: res = "seven" + case: res = "other" + } + return +} + +@(require_results) +shared_target :: proc(x: u8) -> int { + r := 0 + switch x { + case 3, 9: r = 10 + case 4: r = 20 + } + return r + int(x) +} + +@(require_results) +both :: proc(a, b: bool) -> bool { + return a && b +} + +@(require_results) +either :: proc(a, b: bool) -> bool { + return a || b +} + +@(require_results) +count_true :: proc(xs: ..bool) -> (n: int) { + for x in xs { + n += int(x) + } + return +} + +global_pair: Pair + +@(test) +test_fast_isel_lowering :: proc(t: ^testing.T) { + p := pick(true, {"a", 1}, {"bb", 2}) + q := pick(false, {"a", 1}, {"bb", 2}) + testing.expect_value(t, p, Pair{"a", 1}) + testing.expect_value(t, q, Pair{"bb", 2}) + + s := "yes" if p.value == 1 else "no" + testing.expect_value(t, s, "yes") + + global_pair = q + testing.expect_value(t, global_pair.name, "bb") + + src := Wide{1, 2, 3, {4, 5, 6}, {"wide", 7}, true} + dst: Wide + w := copy_through(&dst, &src) + testing.expect_value(t, w, src) + testing.expect_value(t, dst, src) + + ws := []Wide{src, {}, src} + ws[1] = ws[0] + testing.expect_value(t, ws[1], src) + + merged: Pair + if w.f { + merged = p + } else { + merged = q + } + testing.expect_value(t, merged, p) + + names := [4]string{"one", "two", "seven", "other"} + for x, i in ([]int{1, 2, 7, 5}) { + testing.expect_value(t, classify(x), names[i]) + } + testing.expect_value(t, shared_target(3), 13) + testing.expect_value(t, shared_target(9), 19) + testing.expect_value(t, shared_target(4), 24) + testing.expect_value(t, shared_target(5), 5) + + a, b := w.a == 1, w.b == 3 + testing.expect_value(t, both(a, b), false) + testing.expect_value(t, either(a, b), true) + testing.expect_value(t, count_true(a, b, a && !b, a || b), 3) + + buf := [8]u8{1, 2, 3, 4, 5, 6, 7, 8} + copy(buf[1:], buf[:4]) + testing.expect_value(t, buf, [8]u8{1, 1, 2, 3, 4, 6, 7, 8}) + small: [3]u8 = {9, 9, 9} + small = {} + testing.expect_value(t, small, [3]u8{}) +}