From ecfd651f95248b50e075abf504d6221f40acada8 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 15:26:49 +0100 Subject: [PATCH 01/20] Use atomics for the flags `name_test` --- tests/core/thread/test_core_thread.odin | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/core/thread/test_core_thread.odin b/tests/core/thread/test_core_thread.odin index e06fcc3cf..56f42bc86 100644 --- a/tests/core/thread/test_core_thread.odin +++ b/tests/core/thread/test_core_thread.odin @@ -59,9 +59,9 @@ name_test :: proc(_t: ^testing.T) { name_test_t = _t t := thread.create_and_start(name = "test_name", fn = proc() { - main_wait = false + intrinsics.atomic_store(&main_wait, false) - for (child_wait) {} + for intrinsics.atomic_load(&child_wait) {} n, err := thread.get_name() defer if err != nil { @@ -73,7 +73,7 @@ name_test :: proc(_t: ^testing.T) { }) defer free(t) - for (main_wait) {} + for intrinsics.atomic_load(&main_wait) {} n, err := thread.get_name(t) defer if err != nil { @@ -82,7 +82,7 @@ name_test :: proc(_t: ^testing.T) { testing.expect(name_test_t, err == nil, "thread name allocation failed") testing.expectf(name_test_t, n == "test_name","thread name on main did not match : got %v", n) - child_wait = false + intrinsics.atomic_store(&child_wait, false) thread.join(t) } From 32f30c3eb351f8a05000eaea4e255d531037306f Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 16:00:23 +0100 Subject: [PATCH 02/20] Tokenize whilst parsing rather than storing every token of a file first --- src/checker.cpp | 4 +- src/error.cpp | 8 +++ src/main.cpp | 29 +------- src/parser.cpp | 186 ++++++++++++++++++++++++++---------------------- src/parser.hpp | 11 +-- 5 files changed, 119 insertions(+), 119 deletions(-) diff --git a/src/checker.cpp b/src/checker.cpp index fd65d42fc..01702b14e 100644 --- a/src/checker.cpp +++ b/src/checker.cpp @@ -7803,9 +7803,7 @@ gb_internal void check_parsed_files(Checker *c) { token.pos.column = 1; if (s->pkg->files.count > 0) { AstFile *f = s->pkg->files[0]; - if (f->tokens.count > 0) { - token = f->tokens[0]; - } + token = f->first_token; } error(token, "Undefined entry point procedure 'main'"); diff --git a/src/error.cpp b/src/error.cpp index 5879fbeec..f19829928 100644 --- a/src/error.cpp +++ b/src/error.cpp @@ -812,6 +812,10 @@ gb_internal void error_no_newline_va(TokenPos const &pos, char const *fmt, va_li gb_internal void syntax_error_va(TokenPos const &pos, TokenPos end, char const *fmt, va_list va) { + if (global_error_mute_depth > 0) { + global_error_mute_count += 1; + return; + } global_error_collector.count.fetch_add(1); mutex_lock(&global_error_collector.mutex); if (global_error_collector.count > MAX_ERROR_COLLECTOR_COUNT()) { @@ -844,6 +848,10 @@ gb_internal void syntax_error_va(TokenPos const &pos, TokenPos end, char const * } gb_internal void syntax_error_with_verbose_va(TokenPos const &pos, TokenPos end, char const *fmt, va_list va) { + if (global_error_mute_depth > 0) { + global_error_mute_count += 1; + return; + } global_error_collector.count.fetch_add(1); mutex_lock(&global_error_collector.mutex); if (global_error_collector.count > MAX_ERROR_COLLECTOR_COUNT()) { diff --git a/src/main.cpp b/src/main.cpp index 17306bfe1..b541dc0aa 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -2390,12 +2390,10 @@ gb_internal void show_timings(Checker *c, Timings *t) { isize files = 0; isize packages = p->packages.count; isize total_file_size = 0; - f64 total_tokenizing_time = 0; f64 total_parsing_time = 0; for (AstPackage *pkg : p->packages) { files += pkg->files.count; for (AstFile *file : pkg->files) { - total_tokenizing_time += file->time_to_tokenize; total_parsing_time += file->time_to_parse; total_file_size += file->tokenizer.end - file->tokenizer.start; } @@ -2419,22 +2417,9 @@ gb_internal void show_timings(Checker *c, Timings *t) { gb_printf_err("Total File Size - %td\n", total_file_size); gb_printf_err("\n"); } - { - f64 time = total_tokenizing_time; - gb_printf_err("Tokenization Only\n"); - gb_printf_err("LOC/s - %.3f\n", cast(f64)lines/time); - gb_printf_err("us/LOC - %.3f\n", 1.0e6*time/cast(f64)lines); - gb_printf_err("Tokens/s - %.3f\n", cast(f64)tokens/time); - gb_printf_err("us/Token - %.3f\n", 1.0e6*time/cast(f64)tokens); - gb_printf_err("bytes/s - %.3f\n", cast(f64)total_file_size/time); - gb_printf_err("MiB/s - %.3f\n", cast(f64)(total_file_size/time)/(1024*1024)); - gb_printf_err("us/bytes - %.3f\n", 1.0e6*time/cast(f64)total_file_size); - - gb_printf_err("\n"); - } { f64 time = total_parsing_time; - gb_printf_err("Parsing Only\n"); + gb_printf_err("Tokenizing and Parsing Only\n"); gb_printf_err("LOC/s - %.3f\n", cast(f64)lines/time); gb_printf_err("us/LOC - %.3f\n", 1.0e6*time/cast(f64)lines); gb_printf_err("Tokens/s - %.3f\n", cast(f64)tokens/time); @@ -3631,7 +3616,7 @@ gb_internal gbFileError write_file_with_stripped_tokens(gbFile *f, AstFile *file u8 const *file_data = file->tokenizer.start; i32 prev_offset = 0; i32 const end_offset = cast(i32)(file->tokenizer.end - file->tokenizer.start); - for (Token const &token : file->tokens) { + for (Token const &token : file->token_edits) { if (token.flags & (TokenFlag_Remove|TokenFlag_Replace)) { i32 offset = token.pos.offset; i32 to_write = offset-prev_offset; @@ -3674,15 +3659,7 @@ gb_internal int strip_semicolons(Parser *parser) { for (AstPackage *pkg : parser->packages) { for (AstFile *file : pkg->files) { - bool nothing_to_change = true; - for (Token const &token : file->tokens) { - if (token.flags) { - nothing_to_change = false; - break; - } - } - - if (nothing_to_change) { + if (file->token_edits.count == 0) { continue; } diff --git a/src/parser.cpp b/src/parser.cpp index 3f6578ab4..31561f39c 100644 --- a/src/parser.cpp +++ b/src/parser.cpp @@ -1575,13 +1575,35 @@ gb_internal Ast *ast_attribute(AstFile *f, Token token, Token open, Token close, } -gb_internal bool next_token0(AstFile *f) { - if (f->curr_token_index+1 < f->tokens.count) { - f->curr_token = f->tokens[++f->curr_token_index]; - return true; +gb_internal Token tokenize_next(AstFile *f) { + Token token = {}; + tokenizer_get_token(&f->tokenizer, &token); + f->token_count += 1; + if (token.kind == Token_Invalid) { + f->invalid_token_pos = token.pos; + syntax_error(token.pos, "Failed to parse file: %.*s; invalid token found in file", LIT(f->fullpath)); + begin_error_mute(); + token.kind = Token_EOF; } - syntax_error(f->curr_token, "Token is EOF"); - return false; + return token; +} + +gb_internal bool next_token0(AstFile *f) { + if (f->curr_token.kind == Token_EOF) { + syntax_error(f->curr_token, "Token is EOF"); + return false; + } + f->curr_token_index += 1; + if (f->lookahead_index < f->lookahead.count) { + f->curr_token = f->lookahead[f->lookahead_index++]; + if (f->lookahead_index == f->lookahead.count) { + array_clear(&f->lookahead); + f->lookahead_index = 0; + } + } else { + f->curr_token = tokenize_next(f); + } + return true; } @@ -1665,7 +1687,6 @@ gb_internal Token advance_token(AstFile *f) { f->lead_comment = nullptr; f->line_comment = nullptr; - f->prev_token_index = f->curr_token_index; Token prev = f->prev_token = f->curr_token; bool ok = next_token0(f); @@ -1688,30 +1709,40 @@ gb_internal Token advance_token(AstFile *f) { } -gb_internal Token peek_token(AstFile *f) { - for (isize i = f->curr_token_index+1; i < f->tokens.count; i++) { - Token tok = f->tokens[i]; +gb_internal Token peek_token_n(AstFile *f, isize n) { + if (f->curr_token.kind == Token_EOF) { + return {}; + } + for (isize i = f->lookahead_index; /**/; i++) { + if (i == f->lookahead.count) { + array_add(&f->lookahead, tokenize_next(f)); + } + Token tok = f->lookahead[i]; if (tok.kind == Token_Comment) { continue; } - return tok; + if (n-- == 0) { + return tok; + } + if (tok.kind == Token_EOF) { + return {}; + } } - return {}; } -gb_internal Token peek_token_n(AstFile *f, isize n) { - Token found = {}; - for (isize i = f->curr_token_index+1; i < f->tokens.count; i++) { - Token tok = f->tokens[i]; - if (tok.kind == Token_Comment) { - continue; - } - found = tok; - if (n-- == 0) { - return found; - } +gb_internal Token peek_token(AstFile *f) { + return peek_token_n(f, 0); +} + +// the next token, even if it is a comment +gb_internal Token peek_raw_token(AstFile *f) { + if (f->curr_token.kind == Token_EOF) { + return f->curr_token; } - return {}; + if (f->lookahead_index == f->lookahead.count) { + array_add(&f->lookahead, tokenize_next(f)); + } + return f->lookahead[f->lookahead_index]; } @@ -1778,6 +1809,9 @@ gb_internal Token expect_token(AstFile *f, TokenKind kind) { end_error_block(); if (prev.kind == Token_EOF) { + if (f->invalid_token_pos.line != 0) { + end_error_mute(); + } exit_with_errors(); } } @@ -1827,6 +1861,18 @@ gb_internal bool is_token_range(Token tok) { } +gb_internal void add_token_edit(AstFile *f, Token token, u8 flag) { + if (f->token_edits.count > 0) { + Token *last = &f->token_edits[f->token_edits.count-1]; + if (last->pos.offset == token.pos.offset) { + last->flags |= flag; + return; + } + } + token.flags |= flag; + array_add(&f->token_edits, token); +} + gb_internal Token expect_operator(AstFile *f) { Token prev = f->curr_token; if ((prev.kind == Token_in || prev.kind == Token_not_in) && (f->expr_level >= 0 || f->allow_in_expr)) { @@ -1847,7 +1893,7 @@ gb_internal Token expect_operator(AstFile *f) { } if (prev.kind == Token_Ellipsis) { syntax_error(prev, "'..' for ranges are not allowed, did you mean '..<' or '..='?"); - f->tokens[f->curr_token_index].flags |= TokenFlag_Replace; + add_token_edit(f, prev, TokenFlag_Replace); } advance_token(f); @@ -1962,8 +2008,8 @@ gb_internal Token expect_closing(AstFile *f, TokenKind kind, String const &conte gb_internal void assign_removal_flag_to_semicolon(AstFile *f) { // NOTE(bill): this is used for rewriting files to strip unneeded semicolons - Token *prev_token = &f->tokens[f->prev_token_index]; - Token *curr_token = &f->tokens[f->curr_token_index]; + Token const *prev_token = &f->prev_token; + Token const *curr_token = &f->curr_token; GB_ASSERT(prev_token->kind == Token_Semicolon); if (prev_token->string != ";") { return; @@ -1987,7 +2033,7 @@ gb_internal void assign_removal_flag_to_semicolon(AstFile *f) { if (is_strict_style(f) || (ast_file_vet_flags(f) & VetFlag_Semicolon)) { syntax_error(*prev_token, "Found unneeded semicolon"); } - prev_token->flags |= TokenFlag_Remove; + add_token_edit(f, *prev_token, TokenFlag_Remove); } gb_internal void expect_semicolon(AstFile *f) { @@ -2262,11 +2308,7 @@ gb_internal Ast *convert_stmt_to_expr(AstFile *f, Ast *statement, String const & } syntax_error(f->curr_token, "Expected '%.*s', found a simple statement.", LIT(kind)); - Token end = f->curr_token; - if (f->tokens.count < f->curr_token_index) { - end = f->tokens[f->curr_token_index+1]; - } - return ast_bad_expr(f, f->curr_token, end); + return ast_bad_expr(f, f->curr_token, f->curr_token); } gb_internal Ast *convert_stmt_to_body(AstFile *f, Ast *stmt) { @@ -5471,7 +5513,7 @@ if_else_chain:; break; default: syntax_error(f->curr_token, "Expected if statement block statement"); - else_stmt = ast_bad_stmt(f, f->curr_token, f->tokens[f->curr_token_index+1]); + else_stmt = ast_bad_stmt(f, f->curr_token, peek_raw_token(f)); break; } } @@ -5529,7 +5571,7 @@ gb_internal Ast *parse_when_stmt(AstFile *f) { } break; default: syntax_error(f->curr_token, "Expected when statement block statement"); - else_stmt = ast_bad_stmt(f, f->curr_token, f->tokens[f->curr_token_index+1]); + else_stmt = ast_bad_stmt(f, f->curr_token, peek_raw_token(f)); break; } } @@ -6357,7 +6399,7 @@ gb_internal Array parse_stmt_list(AstFile *f) { } -gb_internal ParseFileError init_ast_file(AstFile *f, String const &fullpath, TokenPos *err_pos) { +gb_internal ParseFileError init_ast_file(AstFile *f, String const &fullpath) { GB_ASSERT(f != nullptr); f->fullpath = string_trim_whitespace(fullpath); // Just in case f->filename = remove_directory_from_path(f->fullpath); @@ -6387,51 +6429,23 @@ gb_internal ParseFileError init_ast_file(AstFile *f, String const &fullpath, Tok } - isize file_size = f->tokenizer.end - f->tokenizer.start; - - // NOTE(bill): Determine allocation size required for tokens - isize token_cap = file_size/3ll; - isize pow2_cap = gb_max(cast(isize)prev_pow2(cast(i64)token_cap)/2, 16); - token_cap = ((token_cap + pow2_cap-1)/pow2_cap) * pow2_cap; - - isize init_token_cap = gb_max(token_cap, 16); - array_init(&f->tokens, ast_allocator(f), 0, gb_max(init_token_cap, 16)); + array_init(&f->lookahead, ast_allocator(f), 0, 4); + array_init(&f->token_edits, ast_allocator(f), 0, 0); + array_init(&f->comments, ast_allocator(f), 0, 0); + array_init(&f->imports, ast_allocator(f), 0, 0); if (err == TokenizerInit_Empty) { Token token = {Token_EOF}; token.pos.file_id = f->id; token.pos.line = 1; token.pos.column = 1; - array_add(&f->tokens, token); + f->token_count = 1; + f->first_token = f->prev_token = f->curr_token = token; return ParseFile_None; } - u64 start = time_stamp_time_now(); - - for (;;) { - Token *token = array_add_and_get(&f->tokens); - tokenizer_get_token(&f->tokenizer, token); - if (token->kind == Token_Invalid) { - err_pos->line = token->pos.line; - err_pos->column = token->pos.column; - return ParseFile_InvalidToken; - } - - if (token->kind == Token_EOF) { - break; - } - } - - u64 end = time_stamp_time_now(); - f->time_to_tokenize = cast(f64)(end-start)/cast(f64)time_stamp__freq(); - - f->prev_token_index = 0; f->curr_token_index = 0; - f->prev_token = f->tokens[f->prev_token_index]; - f->curr_token = f->tokens[f->curr_token_index]; - - array_init(&f->comments, ast_allocator(f), 0, 0); - array_init(&f->imports, ast_allocator(f), 0, 0); + f->first_token = f->prev_token = f->curr_token = tokenize_next(f); f->curr_proc = nullptr; @@ -6440,7 +6454,8 @@ gb_internal ParseFileError init_ast_file(AstFile *f, String const &fullpath, Tok gb_internal void destroy_ast_file(AstFile *f) { GB_ASSERT(f != nullptr); - array_free(&f->tokens); + array_free(&f->lookahead); + array_free(&f->token_edits); array_free(&f->comments); array_free(&f->imports); } @@ -7473,11 +7488,8 @@ gb_internal bool parse_file_tag(const String &lc, const Token &tok, AstFile *f) } gb_internal bool parse_file(Parser *p, AstFile *f) { - if (f->tokens.count == 0) { - return true; - } - if (f->tokens.count > 0 && f->tokens[0].kind == Token_EOF) { - return true; + if (f->first_token.kind == Token_EOF) { + return f->invalid_token_pos.line == 0; } u64 start = time_stamp_time_now(); @@ -7604,9 +7616,7 @@ gb_internal ParseFileError process_imported_file(Parser *p, ImportedFile importe AstFile *file = permanent_alloc_item(); file->pkg = pkg; file->id = cast(i32)(imported_file.index+1); - TokenPos err_pos = {0}; - ParseFileError err = init_ast_file(file, fi.fullpath, &err_pos); - err_pos.file_id = file->id; + ParseFileError err = init_ast_file(file, fi.fullpath); file->last_error = err; if (err != ParseFile_None) { @@ -7629,9 +7639,6 @@ gb_internal ParseFileError process_imported_file(Parser *p, ImportedFile importe case ParseFile_NotFound: syntax_error(pos, "Failed to parse file: %.*s; file cannot be found ('%.*s')", LIT(fi.name), LIT(fi.fullpath)); break; - case ParseFile_InvalidToken: - syntax_error(err_pos, "Failed to parse file: %.*s; invalid token found in file", LIT(fi.name)); - break; case ParseFile_EmptyFile: syntax_error(pos, "Failed to parse file: %.*s; file contains no tokens", LIT(fi.name)); break; @@ -7660,7 +7667,14 @@ gb_internal ParseFileError process_imported_file(Parser *p, ImportedFile importe } - if (parse_file(p, file)) { + bool parsed = parse_file(p, file); + if (file->invalid_token_pos.line != 0) { + end_error_mute(); + file->last_error = ParseFile_InvalidToken; + return ParseFile_InvalidToken; + } + + if (parsed) { MUTEX_GUARD_BLOCK(&pkg->files_mutex) { array_add(&pkg->files, file); } @@ -7669,7 +7683,7 @@ gb_internal ParseFileError process_imported_file(Parser *p, ImportedFile importe if (pkg->name.len == 0) { pkg->name = file->package_name; } else if (pkg->name != file->package_name) { - if (file->tokens.count > 0 && file->tokens[0].kind != Token_EOF) { + if (file->first_token.kind != Token_EOF) { Token tok = file->package_token; tok.pos.file_id = file->id; tok.pos.line = gb_max(tok.pos.line, 1); @@ -7683,7 +7697,7 @@ gb_internal ParseFileError process_imported_file(Parser *p, ImportedFile importe mutex_unlock(&pkg->name_mutex); p->total_line_count.fetch_add(file->tokenizer.line_count); - p->total_token_count.fetch_add(file->tokens.count); + p->total_token_count.fetch_add(file->token_count); } return ParseFile_None; diff --git a/src/parser.hpp b/src/parser.hpp index d1988f89b..1f5869883 100644 --- a/src/parser.hpp +++ b/src/parser.hpp @@ -118,11 +118,15 @@ struct AstFile { String directory; Tokenizer tokenizer; - Array tokens; + Array lookahead; // read by peeking, before the parser reaches them + isize lookahead_index; + isize token_count; isize curr_token_index; - isize prev_token_index; + Token first_token; Token curr_token; Token prev_token; // previous non-comment + TokenPos invalid_token_pos; + Array token_edits; // for `-strip-semicolon` Token package_token; String package_name; @@ -151,8 +155,7 @@ struct AstFile { Ast * curr_proc; isize error_count; ParseFileError last_error; - f64 time_to_tokenize; // seconds - f64 time_to_parse; // seconds + f64 time_to_parse; // seconds, including tokenizing CommentGroup *lead_comment; // Comment (block) before the decl CommentGroup *line_comment; // Comment after the semicolon From 3bd53e2a80c2ed728276b5696a9655ed0878f56c Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 16:26:58 +0100 Subject: [PATCH 03/20] Improve the parsing for tokenizing + parsing speeds under `-show-more-timings -show-debug-messages` --- src/main.cpp | 164 +++++++++++++++++++++++++++++++++++++----------- src/parser.cpp | 28 ++++++++- src/parser.hpp | 10 ++- src/timings.cpp | 35 +++++++++++ 4 files changed, 197 insertions(+), 40 deletions(-) diff --git a/src/main.cpp b/src/main.cpp index b541dc0aa..29fcfc899 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -2383,6 +2383,132 @@ gb_internal void show_import_graph(Checker *c) { gb_printf("}\n\n"); } +gb_internal void add_time_to_tokenize_only(AstFile *f, f64 *time, u64 *cpu_time) { + isize size = f->tokenizer.end - f->tokenizer.start; + if (size <= 0) { + return; + } + Tokenizer t = {}; + t.curr_file_id = f->id; + init_tokenizer_with_data(&t, f->fullpath, f->tokenizer.start, size); + + u64 start = time_stamp_time_now(); + u64 cpu_start = thread_cpu_time_now(); + Token token = {}; + do { + tokenizer_get_token(&t, &token); + } while (token.kind != Token_EOF && token.kind != Token_Invalid); + *cpu_time += thread_cpu_time_now()-cpu_start; + *time += cast(f64)(time_stamp_time_now()-start)/cast(f64)time_stamp__freq(); +} + +gb_internal GB_COMPARE_PROC(file_cpu_time_to_parse_cmp) { + AstFile *x = *(AstFile **)a; + AstFile *y = *(AstFile **)b; + if (x->cpu_time_to_parse != y->cpu_time_to_parse) { + return x->cpu_time_to_parse > y->cpu_time_to_parse ? -1 : +1; + } + return string_compare(x->fullpath, y->fullpath); +} + +gb_internal void show_parse_timings(Parser *p, Timings *t) { + auto all_files = array_make(heap_allocator(), 0, p->packages.count*8); + defer (array_free(&all_files)); + for (AstPackage *pkg : p->packages) { + for (AstFile *file : pkg->files) { + array_add(&all_files, file); + } + } + + f64 cpu_freq = thread_cpu_time_freq(); + + isize tokens = p->total_token_count; + isize lines = p->total_line_count; + isize total_file_size = 0; + + f64 load_time = 0; + f64 load_cpu_time = 0; + f64 parse_time = 0; + f64 parse_cpu_time = 0; + f64 setup_time = 0; + f64 setup_cpu_time = 0; + f64 tokenize_time = 0; + u64 tokenize_cpu_ticks = 0; + + for (AstFile *file : all_files) { + total_file_size += file->tokenizer.end - file->tokenizer.start; + load_time += file->time_to_load; + parse_time += file->time_to_parse; + setup_time += file->time_to_setup_decls; + load_cpu_time += cast(f64)file->cpu_time_to_load/cpu_freq; + parse_cpu_time += cast(f64)file->cpu_time_to_parse/cpu_freq; + setup_cpu_time += cast(f64)file->cpu_time_to_setup_decls/cpu_freq; + + add_time_to_tokenize_only(file, &tokenize_time, &tokenize_cpu_ticks); + } + + f64 tokenize_cpu_time = cast(f64)tokenize_cpu_ticks/cpu_freq; + + f64 wall_time = 0; + for (TimeStamp const &s : t->sections) { + if (s.label == "parse files") { + wall_time = time_stamp_as_s(s, t->freq); + break; + } + } + + isize thread_count = gb_max(global_thread_pool.threads.count, 1); + f64 task_time = load_time + parse_time + setup_time; + f64 task_cpu_time = load_cpu_time + parse_cpu_time + setup_cpu_time; + f64 mib = cast(f64)total_file_size/(1024.0*1024.0); + + auto const &row = [&](char const *label, f64 time, f64 cpu_time, bool rates) { + gb_printf_err(" %s - %9.3f ms %9.3f ms", label, 1.0e3*time, 1.0e3*cpu_time); + if (rates) { + f64 lines_per_second = cast(f64)lines/cpu_time; + char const *loc = " LOC/s"; + if (lines_per_second >= 1.0e6) { + lines_per_second /= 1.0e6; + loc = " MLOC/s"; + } else if (lines_per_second >= 1.0e3) { + lines_per_second /= 1.0e3; + loc = " kLOC/s"; + } + gb_printf_err(" %7.3f us/token %7.3f us/line %8.2f MiB/s %7.2f%s", + 1.0e6*cpu_time/cast(f64)tokens, 1.0e6*cpu_time/cast(f64)lines, mib/cpu_time, lines_per_second, loc); + } + gb_printf_err("\n"); + }; + + gb_printf_err("Parsing (%td files, %td lines, %td tokens, %.2f MiB, %td threads)\n", all_files.count, lines, tokens, mib, thread_count); + gb_printf_err(" parse files - %9.3f ms (wall time)\n", 1.0e3*wall_time); + gb_printf_err(" wall time CPU time (rates from the CPU time)\n"); + row( "file tasks ", task_time, task_cpu_time, false); + row( " loading files ", load_time, load_cpu_time, false); + row( " tokenizing and parsing ", parse_time, parse_cpu_time, true); + row( " setting up decls and imports ", setup_time, setup_cpu_time, false); + row( "tokenizing alone, 1 thread ", tokenize_time, tokenize_cpu_time, true); + if (build_context.thread_count == 1) { + row( "parsing alone (estimated) ", parse_time-tokenize_time, parse_cpu_time-tokenize_cpu_time, true); + } else { + gb_printf_err(" parsing alone - estimated with -thread-count:1\n"); + } + gb_printf_err(" the file tasks kept the threads %.0f%% busy and %.0f%% of their time was spent running\n", + 100.0*task_time/(wall_time*cast(f64)thread_count), 100.0*task_cpu_time/task_time); + gb_printf_err("\n"); + + array_sort(all_files, file_cpu_time_to_parse_cmp); + gb_printf_err("Slowest files to tokenize and parse, by CPU time\n"); + for (isize i = 0; i < gb_min(all_files.count, 8); i++) { + AstFile *file = all_files[i]; + f64 cpu_time = cast(f64)file->cpu_time_to_parse/cpu_freq; + gb_printf_err(" %9.3f ms %9.3f ms (wall) %8td tokens %7.3f us/token %.*s\n", + 1.0e3*cpu_time, 1.0e3*file->time_to_parse, file->token_count, + 1.0e6*cpu_time/cast(f64)gb_max(file->token_count, 1), LIT(file->fullpath)); + } + gb_printf_err("\n"); +} + gb_internal void show_timings(Checker *c, Timings *t) { Parser *p = c->parser; isize lines = p->total_line_count; @@ -2390,11 +2516,9 @@ gb_internal void show_timings(Checker *c, Timings *t) { isize files = 0; isize packages = p->packages.count; isize total_file_size = 0; - f64 total_parsing_time = 0; for (AstPackage *pkg : p->packages) { files += pkg->files.count; for (AstFile *file : pkg->files) { - total_parsing_time += file->time_to_parse; total_file_size += file->tokenizer.end - file->tokenizer.start; } } @@ -2417,41 +2541,7 @@ gb_internal void show_timings(Checker *c, Timings *t) { gb_printf_err("Total File Size - %td\n", total_file_size); gb_printf_err("\n"); } - { - f64 time = total_parsing_time; - gb_printf_err("Tokenizing and Parsing Only\n"); - gb_printf_err("LOC/s - %.3f\n", cast(f64)lines/time); - gb_printf_err("us/LOC - %.3f\n", 1.0e6*time/cast(f64)lines); - gb_printf_err("Tokens/s - %.3f\n", cast(f64)tokens/time); - gb_printf_err("us/Token - %.3f\n", 1.0e6*time/cast(f64)tokens); - gb_printf_err("bytes/s - %.3f\n", cast(f64)total_file_size/time); - gb_printf_err("MiB/s - %.3f\n", cast(f64)(total_file_size/time)/(1024*1024)); - gb_printf_err("us/bytes - %.3f\n", 1.0e6*time/cast(f64)total_file_size); - - gb_printf_err("\n"); - } - { - TimeStamp ts = {}; - for (TimeStamp const &s : t->sections) { - if (s.label == "parse files") { - ts = s; - break; - } - } - GB_ASSERT(ts.label == "parse files"); - - f64 parse_time = time_stamp_as_s(ts, t->freq); - gb_printf_err("Parse pass\n"); - gb_printf_err("LOC/s - %.3f\n", cast(f64)lines/parse_time); - gb_printf_err("us/LOC - %.3f\n", 1.0e6*parse_time/cast(f64)lines); - gb_printf_err("Tokens/s - %.3f\n", cast(f64)tokens/parse_time); - gb_printf_err("us/Token - %.3f\n", 1.0e6*parse_time/cast(f64)tokens); - gb_printf_err("bytes/s - %.3f\n", cast(f64)total_file_size/parse_time); - gb_printf_err("MiB/s - %.3f\n", cast(f64)(total_file_size/parse_time)/(1024*1024)); - gb_printf_err("us/bytes - %.3f\n", 1.0e6*parse_time/cast(f64)total_file_size); - - gb_printf_err("\n"); - } + show_parse_timings(p, t); { TimeStamp ts = {}; TimeStamp ts_end = {}; diff --git a/src/parser.cpp b/src/parser.cpp index 31561f39c..5b8329ba6 100644 --- a/src/parser.cpp +++ b/src/parser.cpp @@ -6399,6 +6399,15 @@ gb_internal Array parse_stmt_list(AstFile *f) { } +// Only the report of `-show-more-timings -show-debug-messages` uses a file's CPU time, +// and getting a thread's CPU time is a system call +gb_internal u64 parse_thread_cpu_time_now(void) { + if (build_context.show_debug_messages && build_context.show_more_timings) { + return thread_cpu_time_now(); + } + return 0; +} + gb_internal ParseFileError init_ast_file(AstFile *f, String const &fullpath) { GB_ASSERT(f != nullptr); f->fullpath = string_trim_whitespace(fullpath); // Just in case @@ -6412,7 +6421,11 @@ gb_internal ParseFileError init_ast_file(AstFile *f, String const &fullpath) { gb_zero_item(&f->tokenizer); f->tokenizer.curr_file_id = f->id; + u64 load_start = time_stamp_time_now(); + u64 load_cpu_start = parse_thread_cpu_time_now(); TokenizerInitError err = init_tokenizer_from_fullpath(&f->tokenizer, f->fullpath, build_context.copy_file_contents); + f->cpu_time_to_load = parse_thread_cpu_time_now()-load_cpu_start; + f->time_to_load = cast(f64)(time_stamp_time_now()-load_start)/cast(f64)time_stamp__freq(); if (err != TokenizerInit_None) { switch (err) { case TokenizerInit_Empty: @@ -7493,6 +7506,9 @@ gb_internal bool parse_file(Parser *p, AstFile *f) { } u64 start = time_stamp_time_now(); + u64 cpu_start = parse_thread_cpu_time_now(); + u64 setup_start = 0; + u64 setup_cpu_start = 0; String filepath = f->tokenizer.fullpath; String base_dir = dir_from_path(filepath); @@ -7593,11 +7609,19 @@ gb_internal bool parse_file(Parser *p, AstFile *f) { f->decls = slice_from_array(decls); + setup_start = time_stamp_time_now(); + setup_cpu_start = parse_thread_cpu_time_now(); parse_setup_file_decls(p, f, base_dir, f->decls); } - u64 end = time_stamp_time_now(); - f->time_to_parse = cast(f64)(end-start)/cast(f64)time_stamp__freq(); + u64 end = time_stamp_time_now(); + u64 cpu_end = parse_thread_cpu_time_now(); + u64 setup_ticks = setup_start != 0 ? end-setup_start : 0; + u64 setup_cpu_ticks = setup_cpu_start != 0 ? cpu_end-setup_cpu_start : 0; + f->time_to_parse = cast(f64)(end-start-setup_ticks)/cast(f64)time_stamp__freq(); + f->time_to_setup_decls = cast(f64)setup_ticks/cast(f64)time_stamp__freq(); + f->cpu_time_to_parse = cpu_end-cpu_start-setup_cpu_ticks; + f->cpu_time_to_setup_decls = setup_cpu_ticks; for (int i = 0; i < AstDelayQueue_COUNT; i++) { array_init(f->delayed_decls_queues+i, ast_allocator(f), 0, f->delayed_decl_count); diff --git a/src/parser.hpp b/src/parser.hpp index 1f5869883..65d0cde3a 100644 --- a/src/parser.hpp +++ b/src/parser.hpp @@ -155,7 +155,6 @@ struct AstFile { Ast * curr_proc; isize error_count; ParseFileError last_error; - f64 time_to_parse; // seconds, including tokenizing CommentGroup *lead_comment; // Comment (block) before the decl CommentGroup *line_comment; // Comment after the semicolon @@ -173,6 +172,15 @@ struct AstFile { struct LLVMOpaqueMetadata *llvm_metadata; struct LLVMOpaqueMetadata *llvm_metadata_scope; + + //// Profiling ///// + + f64 time_to_load; // seconds + f64 time_to_parse; // seconds, tokenizing included, setting up the decls excluded + f64 time_to_setup_decls; // seconds, mostly finding and adding the imported packages + u64 cpu_time_to_load; + u64 cpu_time_to_parse; + u64 cpu_time_to_setup_decls; }; enum AstForeignFileKind { diff --git a/src/timings.cpp b/src/timings.cpp index f6e86867f..f1d6e5eb3 100644 --- a/src/timings.cpp +++ b/src/timings.cpp @@ -32,6 +32,7 @@ gb_internal u64 win32_time_stamp__freq(void) { #elif defined(GB_SYSTEM_OSX) #include +#include gb_internal mach_timebase_info_data_t osx_init_timebase_info(void) { mach_timebase_info_data_t data; @@ -105,6 +106,40 @@ gb_internal u64 time_stamp__freq(void) { #endif } +gb_internal u64 thread_cpu_time_now(void) { +#if defined(GB_SYSTEM_WINDOWS) + ULONG64 cycles = 0; + QueryThreadCycleTime(GetCurrentThread(), &cycles); + return cycles; +#else + struct timespec ts; + clock_gettime(CLOCK_THREAD_CPUTIME_ID, &ts); + return (cast(u64)ts.tv_sec * 1000000000ull) + cast(u64)ts.tv_nsec; +#endif +} + +gb_internal f64 thread_cpu_time_freq(void) { +#if defined(GB_SYSTEM_WINDOWS) + gb_local_persist f64 freq = 0; + if (freq == 0) { + for (isize i = 0; i < 5; i++) { + u64 start = time_stamp_time_now(); + u64 start_cycles = thread_cpu_time_now(); + u64 end = start; + while (end-start < time_stamp__freq()/100) { + end = time_stamp_time_now(); + } + u64 end_cycles = thread_cpu_time_now(); + f64 measured = cast(f64)(end_cycles-start_cycles) * cast(f64)time_stamp__freq() / cast(f64)(end-start); + freq = gb_max(freq, measured); + } + } + return freq; +#else + return 1.0e9; +#endif +} + gb_internal TimeStamp make_time_stamp(String const &label) { TimeStamp ts = {0}; ts.start = time_stamp_time_now(); From 27e12a53cd19ad1649259a6d1d1500f6fd2e3399 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 17:18:38 +0100 Subject: [PATCH 04/20] Inline the ASCII paths of `advance_to_next_rune` --- src/tokenizer.cpp | 28 ++++++++++++++++++++-------- src/unicode.cpp | 34 ++++++++++++++-------------------- 2 files changed, 34 insertions(+), 28 deletions(-) diff --git a/src/tokenizer.cpp b/src/tokenizer.cpp index 84d4d3041..d3bf23519 100644 --- a/src/tokenizer.cpp +++ b/src/tokenizer.cpp @@ -348,18 +348,14 @@ gb_internal void tokenizer_err(Tokenizer *t, TokenPos const &pos, char const *ms t->error_count++; } -gb_internal void advance_to_next_rune(Tokenizer *t) { - if (t->curr_rune == '\n') { - t->column_minus_one = -1; - t->line_count++; - } +gb_internal void advance_to_next_rune_slow(Tokenizer *t) { if (t->read_curr < t->end) { t->curr = t->read_curr; Rune rune = *t->read_curr; if (rune == 0) { tokenizer_err(t, "Illegal character NUL"); t->read_curr++; - } else if (rune & 0x80) { // not ASCII + } else { // not ASCII isize width = utf8_decode(t->read_curr, t->end-t->read_curr, &rune); t->read_curr += width; if (rune == GB_RUNE_INVALID && width == 1) { @@ -367,8 +363,6 @@ gb_internal void advance_to_next_rune(Tokenizer *t) { } else if (rune == GB_RUNE_BOM && t->curr-t->start > 0){ tokenizer_err(t, "Illegal byte order mark"); } - } else { - t->read_curr++; } t->curr_rune = rune; t->column_minus_one++; @@ -378,6 +372,24 @@ gb_internal void advance_to_next_rune(Tokenizer *t) { } } +gb_internal gb_inline void advance_to_next_rune(Tokenizer *t) { + if (t->curr_rune == '\n') { + t->column_minus_one = -1; + t->line_count++; + } + if (t->read_curr < t->end) { + u8 c = *t->read_curr; + if (c != 0 && c < 0x80) { + t->curr = t->read_curr; + t->read_curr++; + t->curr_rune = c; + t->column_minus_one++; + return; + } + } + advance_to_next_rune_slow(t); +} + gb_internal void init_tokenizer_with_data(Tokenizer *t, String const &fullpath, void const *data, isize size) { t->fullpath = fullpath; t->column_minus_one = -1; diff --git a/src/unicode.cpp b/src/unicode.cpp index b38c6d601..98fa27b1c 100644 --- a/src/unicode.cpp +++ b/src/unicode.cpp @@ -12,13 +12,7 @@ extern "C" { #endif -gb_internal bool rune_is_letter(Rune r) { - if (r < 0x80) { - if (r == '_') { - return true; - } - return ((cast(u32)r | 0x20) - 0x61) < 26; - } +gb_internal bool rune_is_letter_non_ascii(Rune r) { switch (utf8proc_category(r)) { case UTF8PROC_CATEGORY_LU: case UTF8PROC_CATEGORY_LL: @@ -30,14 +24,24 @@ gb_internal bool rune_is_letter(Rune r) { return false; } -gb_internal bool rune_is_digit(Rune r) { +gb_internal gb_inline bool rune_is_letter(Rune r) { + if (r < 0x80) { + if (r == '_') { + return true; + } + return ((cast(u32)r | 0x20) - 0x61) < 26; + } + return rune_is_letter_non_ascii(r); +} + +gb_internal gb_inline bool rune_is_digit(Rune r) { if (r < 0x80) { return (cast(u32)r - '0') < 10; } return utf8proc_category(r) == UTF8PROC_CATEGORY_ND; } -gb_internal bool rune_is_letter_or_digit(Rune r) { +gb_internal gb_inline bool rune_is_letter_or_digit(Rune r) { if (r < 0x80) { if (r == '_') { return true; @@ -47,17 +51,7 @@ gb_internal bool rune_is_letter_or_digit(Rune r) { } return (cast(u32)r - '0') < 10; } - switch (utf8proc_category(r)) { - case UTF8PROC_CATEGORY_LU: - case UTF8PROC_CATEGORY_LL: - case UTF8PROC_CATEGORY_LT: - case UTF8PROC_CATEGORY_LM: - case UTF8PROC_CATEGORY_LO: - return true; - case UTF8PROC_CATEGORY_ND: - return true; - } - return false; + return rune_is_letter_non_ascii(r) || utf8proc_category(r) == UTF8PROC_CATEGORY_ND; } gb_internal bool rune_is_whitespace(Rune r) { From 001be12b3440ff9ef145effe03bf5c2f0be8b687 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 17:36:19 +0100 Subject: [PATCH 05/20] Scan the runs of ASCII characters is identifiers, and simplify the hash for the keyword look up --- src/tokenizer.cpp | 52 +++++++++++++++++++++++++++++++++++++++++++---- 1 file changed, 48 insertions(+), 4 deletions(-) diff --git a/src/tokenizer.cpp b/src/tokenizer.cpp index d3bf23519..6900a2098 100644 --- a/src/tokenizer.cpp +++ b/src/tokenizer.cpp @@ -154,16 +154,29 @@ GB_STATIC_ASSERT(Token__KeywordEnd-Token__KeywordBegin <= gb_count_of(keyword_ha gb_global isize const min_keyword_size = 2; gb_global isize max_keyword_size = 11; gb_global bool keyword_indices[16] = {}; +gb_global u32 keyword_first_letters = 0; // a bit for each letter from 'a' which a keyword starts with gb_internal gb_inline u32 keyword_hash(u8 const *text, isize len) { - return fnv32a(text, len); + return cast(u32)len + text[0] + 3*text[1] + 26*text[len-1]; } + +gb_internal gb_inline bool could_be_keyword(String const &s) { + if (s.len < min_keyword_size || s.len > max_keyword_size || !keyword_indices[s.len]) { + return false; + } + u32 first = cast(u32)s[0] - 'a'; + return first < 26 && (keyword_first_letters & (1u< t->read_curr) { + t->column_minus_one += cast(i32)(p - t->read_curr); + t->curr = p-1; + t->curr_rune = p[-1]; + t->read_curr = p; + } +} + gb_internal void init_tokenizer_with_data(Tokenizer *t, String const &fullpath, void const *data, isize size) { t->fullpath = fullpath; t->column_minus_one = -1; @@ -663,10 +686,24 @@ gb_internal bool scan_escape(Tokenizer *t) { gb_internal gb_inline void tokenizer_skip_line(Tokenizer *t) { while (t->curr_rune != '\n' && t->curr_rune != GB_RUNE_EOF) { + u8 *p = t->read_curr; + while (p < t->end && *p != '\n' && *p != 0 && *p < 0x80) { + p++; + } + tokenizer_skip_ascii_to(t, p); advance_to_next_rune(t); } } +gb_internal gb_inline void tokenizer_skip_spaces(Tokenizer *t) { + u8 *p = t->read_curr; + while (p < t->end && (*p == ' ' || *p == '\t' || *p == '\r')) { + p++; + } + tokenizer_skip_ascii_to(t, p); + advance_to_next_rune(t); +} + gb_internal gb_inline void tokenizer_skip_whitespace(Tokenizer *t, bool on_newline) { if (on_newline) { for (;;) { @@ -674,7 +711,7 @@ gb_internal gb_inline void tokenizer_skip_whitespace(Tokenizer *t, bool on_newli case ' ': case '\t': case '\r': - advance_to_next_rune(t); + tokenizer_skip_spaces(t); continue; } break; @@ -683,10 +720,12 @@ gb_internal gb_inline void tokenizer_skip_whitespace(Tokenizer *t, bool on_newli for (;;) { switch (t->curr_rune) { case '\n': + advance_to_next_rune(t); + continue; case ' ': case '\t': case '\r': - advance_to_next_rune(t); + tokenizer_skip_spaces(t); continue; } break; @@ -710,6 +749,11 @@ gb_internal void tokenizer_get_token(Tokenizer *t, Token *token, int repeat=0) { Rune curr_rune = t->curr_rune; if (rune_is_letter(curr_rune)) { token->kind = Token_Ident; + u8 *p = t->read_curr; + while (p < t->end && *p < 0x80 && rune_is_letter_or_digit(*p)) { + p++; + } + tokenizer_skip_ascii_to(t, p); while (rune_is_letter_or_digit(t->curr_rune)) { advance_to_next_rune(t); } @@ -717,7 +761,7 @@ gb_internal void tokenizer_get_token(Tokenizer *t, Token *token, int repeat=0) { token->string.len = t->curr - token->string.text; // NOTE(bill): Heavily optimize to make it faster to find keywords - if (1 < token->string.len && token->string.len <= max_keyword_size && keyword_indices[token->string.len]) { + if (could_be_keyword(token->string)) { u32 hash = keyword_hash(token->string.text, token->string.len); u32 index = hash & KEYWORD_HASH_TABLE_MASK; KeywordHashEntry *entry = &keyword_hash_table[index]; From 5c787d0e729fcfc5a67c780435d94b7a4f5ddcee Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 18:17:25 +0100 Subject: [PATCH 06/20] Re-enable the `temporary_allocator()` and fix the `arena_temp_end` --- src/check_expr.cpp | 1 - src/common_memory.cpp | 18 +++++++++--------- 2 files changed, 9 insertions(+), 10 deletions(-) diff --git a/src/check_expr.cpp b/src/check_expr.cpp index 75190b66a..72f57da35 100644 --- a/src/check_expr.cpp +++ b/src/check_expr.cpp @@ -8897,7 +8897,6 @@ gb_internal CallArgumentError check_polymorphic_record_type(CheckerContext *c, O c->allow_in_progress_type_operand = prev_allow_in_progress; }); - TEMPORARY_ALLOCATOR_GUARD(); if (is_call_expr_field_value(ce)) { named_fields = true; operands = array_make(temporary_allocator(), ce->args.count); diff --git a/src/common_memory.cpp b/src/common_memory.cpp index ac82e824c..8e606a62c 100644 --- a/src/common_memory.cpp +++ b/src/common_memory.cpp @@ -392,12 +392,14 @@ ArenaTemp arena_temp_begin(Arena *arena) { GB_ASSERT(arena); GB_ASSERT(arena->parent_thread == get_current_thread()); + if (arena->curr_block == nullptr) { + arena_alloc(arena, 0, 1); + } + ArenaTemp temp = {}; temp.arena = arena; temp.block = arena->curr_block; - if (arena->curr_block != nullptr) { - temp.used = arena->curr_block->used; - } + temp.used = arena->curr_block->used; arena->temp_count += 1; return temp; } @@ -428,8 +430,8 @@ void arena_temp_end(ArenaTemp const &temp) { MemoryBlock *block = arena->curr_block; if (block) { GB_ASSERT_MSG(block->used >= temp.used, "out of order use of arena_temp_end"); - isize amount_to_zero = gb_min(block->used - temp.used, block->size - block->used); - gb_zero_size(block->base + temp.used, amount_to_zero); + // `arena_alloc` expects the memory to be zeroed already + gb_zero_size(block->base + temp.used, block->used - temp.used); block->used = temp.used; } } @@ -605,16 +607,14 @@ gb_internal gbAllocator permanent_allocator() { } gb_internal gbAllocator temporary_allocator() { - // return {thread_arena_allocator_proc, cast(void *)cast(uintptr)ThreadArena_Temporary}; - return permanent_allocator(); + return {thread_arena_allocator_proc, cast(void *)cast(uintptr)ThreadArena_Temporary}; } #define TEMP_ARENA_GUARD(arena) ArenaTempGuard GB_DEFER_3(_arena_guard_){arena} -// #define TEMPORARY_ALLOCATOR_GUARD() TEMP_ARENA_GUARD(get_arena(ThreadArena_Temporary)) -#define TEMPORARY_ALLOCATOR_GUARD() +#define TEMPORARY_ALLOCATOR_GUARD() TEMP_ARENA_GUARD(get_arena(ThreadArena_Temporary)) #define PERMANENT_ALLOCATOR_GUARD() From cd08004d33f8982477030c090c48d22b4d6e6911 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 18:38:09 +0100 Subject: [PATCH 07/20] Free the temporary memory of each procedure once it is generated; keep the strings used as keys of the constant string maps permanently --- src/llvm_backend.cpp | 2 ++ src/llvm_backend_const.cpp | 3 +-- src/llvm_backend_proc.cpp | 4 ++-- 3 files changed, 5 insertions(+), 4 deletions(-) diff --git a/src/llvm_backend.cpp b/src/llvm_backend.cpp index 8a015669d..900a762f9 100644 --- a/src/llvm_backend.cpp +++ b/src/llvm_backend.cpp @@ -3789,6 +3789,8 @@ gb_internal void lb_generate_procedure(lbModule *m, lbProcedure *p) { return; } + TEMPORARY_ALLOCATOR_GUARD(); + if (p->body != nullptr) { // Build Procedure m->curr_procedure = p; lb_begin_procedure_body(p); diff --git a/src/llvm_backend_const.cpp b/src/llvm_backend_const.cpp index db123b142..7eda9515d 100644 --- a/src/llvm_backend_const.cpp +++ b/src/llvm_backend_const.cpp @@ -1317,8 +1317,7 @@ gb_internal lbValue lb_const_value(lbModule *m, Type *type, ExactValue value, lb isize len = value.value_string.len; if (is_type_string16(res.type) || is_type_cstring16(res.type)) { - TEMPORARY_ALLOCATOR_GUARD(); - String16 s16 = string_to_string16(temporary_allocator(), value.value_string); + String16 s16 = string_to_string16(permanent_allocator(), value.value_string); len = s16.len; ptr = lb_find_or_add_entity_string16_ptr(m, s16, custom_link_section); } else { diff --git a/src/llvm_backend_proc.cpp b/src/llvm_backend_proc.cpp index 3172bc0b0..6a2ac0faa 100644 --- a/src/llvm_backend_proc.cpp +++ b/src/llvm_backend_proc.cpp @@ -4959,7 +4959,7 @@ gb_internal lbValue lb_handle_param_value(lbProcedure *p, Type *parameter_type, { Ast *orig = param_value.original_ast_expr; if (orig->kind == Ast_BasicDirective) { - gbString expr = expr_to_string(call_expression, temporary_allocator()); + gbString expr = expr_to_string(call_expression, permanent_allocator()); return lb_const_string(p->module, make_string_c(expr)); } @@ -4999,7 +4999,7 @@ gb_internal lbValue lb_handle_param_value(lbProcedure *p, Type *parameter_type, } } - gbString expr = expr_to_string(target_expr, temporary_allocator()); + gbString expr = expr_to_string(target_expr, permanent_allocator()); return lb_const_string(p->module, make_string_c(expr)); } From 5795183b757176d42d739570e3616b5b9d1e2020 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 19:37:26 +0100 Subject: [PATCH 08/20] Queue a polymorphic specialization's body when it is first used rather than when it is generated, so the minimum dependency set is only generated once; assert that every body in it was checked rather than checking any missed ones late --- src/check_expr.cpp | 40 ++++--------- src/checker.cpp | 138 ++++++++++++--------------------------------- src/checker.hpp | 1 + 3 files changed, 48 insertions(+), 131 deletions(-) diff --git a/src/check_expr.cpp b/src/check_expr.cpp index 72f57da35..7f079e5bf 100644 --- a/src/check_expr.cpp +++ b/src/check_expr.cpp @@ -568,27 +568,11 @@ gb_internal Type *strip_poly_specialized_proc_type(Type *full) { return t; } -// Reuse an existing generated specialization `other`, scheduling its body if unchecked. -// Caller must have released gen_procs->mutex first. -gb_internal bool reuse_gen_polymorphic_procedure(Checker *checker, Entity *other, Ast *poly_def_node, PolyProcData *poly_proc_data) { +// Reuse an existing generated specialization `other`, whose body is checked once it is used +gb_internal bool reuse_gen_polymorphic_procedure(Entity *other, PolyProcData *poly_proc_data) { if (poly_proc_data) { poly_proc_data->gen_entity = other; } - - DeclInfo *decl = other->decl_info; - if (decl->proc_checked_state != ProcCheckedState_Checked) { - ProcInfo *proc_info = permanent_alloc_item(); - proc_info->file = other->file; - proc_info->token = other->token; - proc_info->decl = decl; - proc_info->type = proc_entity_full_type(other); - proc_info->body = decl->proc_lit->ProcLit.body; - proc_info->tags = other->Procedure.tags; - proc_info->generated_from_polymorphic = true; - proc_info->poly_def_node = poly_def_node; - - check_procedure_later(checker, proc_info); - } return true; } @@ -773,7 +757,7 @@ gb_internal bool find_or_generate_polymorphic_procedure(CheckerContext *old_c, E Type *pt = base_type(proc_entity_full_type(other)); if (are_types_identical(pt, final_proc_type)) { rw_mutex_shared_unlock(&gen_procs->mutex); // @local-mutex - return reuse_gen_polymorphic_procedure(nctx.checker, other, poly_def_node, poly_proc_data); + return reuse_gen_polymorphic_procedure(other, poly_proc_data); } } rw_mutex_shared_unlock(&gen_procs->mutex); // @local-mutex @@ -788,7 +772,7 @@ gb_internal bool find_or_generate_polymorphic_procedure(CheckerContext *old_c, E Entity *other = gen_procs->procs[i]; if (gen_procs->hashes[i] == final_hash && are_types_identical(base_type(proc_entity_full_type(other)), final_proc_type)) { rw_mutex_unlock(&gen_procs->mutex); // @local-mutex - return reuse_gen_polymorphic_procedure(nctx.checker, other, poly_def_node, poly_proc_data); + return reuse_gen_polymorphic_procedure(other, poly_proc_data); } } @@ -860,10 +844,6 @@ gb_internal bool find_or_generate_polymorphic_procedure(CheckerContext *old_c, E AstFile *file = base_entity->file; - array_add(&gen_procs->procs, entity); - array_add(&gen_procs->hashes, proc_type_identity_hash(final_proc_type)); - rw_mutex_unlock(&gen_procs->mutex); // @local-mutex - ProcInfo *proc_info = permanent_alloc_item(); proc_info->file = file; proc_info->token = token; @@ -874,6 +854,12 @@ gb_internal bool find_or_generate_polymorphic_procedure(CheckerContext *old_c, E proc_info->generated_from_polymorphic = true; proc_info->poly_def_node = poly_def_node; + // Before it can be found by another thread which could use it first + d->gen_proc_info.store(proc_info); + + array_add(&gen_procs->procs, entity); + array_add(&gen_procs->hashes, proc_type_identity_hash(final_proc_type)); + rw_mutex_unlock(&gen_procs->mutex); // @local-mutex if (poly_proc_data) { poly_proc_data->gen_entity = entity; @@ -926,9 +912,6 @@ gb_internal bool find_or_generate_polymorphic_procedure(CheckerContext *old_c, E } } - // NOTE(bill): Check the newly generated procedure body - check_procedure_later(nctx.checker, proc_info); - return true; } @@ -7771,9 +7754,6 @@ gb_internal bool check_call_arguments_single(CheckerContext *c, Ast *call, Opera } else { decl->where_clauses_evaluated = true; - if (ok && (data->gen_entity->flags & EntityFlag_ProcBodyChecked) == 0) { - check_procedure_later(c->checker, e->file, e->token, decl, proc_entity_full_type(e), decl->proc_lit->ProcLit.body, decl->proc_lit->ProcLit.tags); - } if (is_type_proc(data->gen_entity->type)) { Type *t = base_type(entity_to_use->type); data->result_type = t->Proc.results; diff --git a/src/checker.cpp b/src/checker.cpp index 01702b14e..0a4dec1fb 100644 --- a/src/checker.cpp +++ b/src/checker.cpp @@ -12,6 +12,7 @@ gb_internal void check_expr(CheckerContext *c, Operand *operand, Ast *expression gb_internal void check_expr_or_type(CheckerContext *c, Operand *operand, Ast *expression, Type *type_hint=nullptr); gb_internal void add_comparison_procedures_for_fields(CheckerContext *c, Type *t); gb_internal Type *check_type(CheckerContext *ctx, Ast *e); +gb_internal void check_procedure_later(Checker *c, ProcInfo *info); gb_internal bool is_operand_value(Operand o) { switch (o.mode) { @@ -2204,6 +2205,16 @@ gb_internal void add_entity_use(CheckerContext *c, Ast *identifier, Entity *enti } add_declaration_dependency(c, entity); entity->flags |= EntityFlag_Used; + if (entity->kind == Entity_Procedure && entity->Procedure.generated_from_polymorphic) { + // Check the specialization body later + DeclInfo *decl = entity->decl_info; + if (decl != nullptr) { + ProcInfo *pi = decl->gen_proc_info.exchange(nullptr); + if (pi != nullptr) { + check_procedure_later(c->checker, pi); + } + } + } if (entity_has_deferred_procedure(entity)) { Entity *deferred = entity->Procedure.deferred_procedure.entity; if (deferred != entity) { @@ -6432,60 +6443,6 @@ gb_internal void calculate_global_init_order(Checker *c) { } } -gb_internal void check_procedure_later_from_entity(Checker *c, Entity *e, char const *from_msg) { - if (e == nullptr || e->kind != Entity_Procedure) { - return; - } - if (e->Procedure.is_foreign) { - return; - } - if ((e->flags & EntityFlag_ProcBodyChecked) != 0) { - return; - } - if ((e->flags & EntityFlag_Overridden) != 0) { - // NOTE (zen3ger) Delay checking of a proc alias until the underlying proc is checked. - GB_ASSERT(e->aliased_of != nullptr); - GB_ASSERT(e->aliased_of->kind == Entity_Procedure); - if ((e->aliased_of->flags & EntityFlag_ProcBodyChecked) != 0) { - e->flags |= EntityFlag_ProcBodyChecked; - return; - } - // NOTE (zen3ger) A proc alias *does not* have a body and tags! - check_procedure_later(c, e->file, e->token, e->decl_info, e->type, nullptr, 0); - return; - } - Type *type = base_type(e->type); - if (type == t_invalid) { - return; - } - GB_ASSERT_MSG(type->kind == Type_Proc, "%s", type_to_string(e->type)); - - if (is_type_polymorphic(type) && !type->Proc.is_poly_specialized) { - return; - } - - GB_ASSERT(e->decl_info != nullptr); - - ProcInfo *pi = permanent_alloc_item(); - pi->file = e->file; - pi->token = e->token; - pi->decl = e->decl_info; - pi->type = proc_entity_full_type(e); - - Ast *pl = e->decl_info->proc_lit; - GB_ASSERT(pl != nullptr); - pi->body = pl->ProcLit.body; - pi->tags = pl->ProcLit.tags; - if (pi->body == nullptr) { - return; - } - if (from_msg != nullptr) { - debugf("CHECK PROCEDURE LATER [FROM %s]! %.*s :: %s {...}\n", from_msg, LIT(e->token.string), type_to_string(e->type)); - } - check_procedure_later(c, pi); -} - - gb_internal bool check_proc_info(Checker *c, ProcInfo *pi, UntypedExprInfoMap *untyped) { if (pi == nullptr) { return false; @@ -6534,12 +6491,7 @@ gb_internal bool check_proc_info(Checker *c, ProcInfo *pi, UntypedExprInfoMap *u if (pt->is_polymorphic && pt->is_poly_specialized) { Entity *e = pi->decl->entity; GB_ASSERT(e != nullptr); - if ((e->flags & EntityFlag_Used) == 0) { - // NOTE(bill, 2019-08-31): It was never used, don't check - // NOTE(bill, 2023-01-02): This may need to be checked again if it is used elsewhere? - pi->decl->proc_checked_state.store(ProcCheckedState_Unchecked); - return false; - } + GB_ASSERT_MSG((e->flags & EntityFlag_Used) != 0, "unused specialization '%.*s' queued for checking", LIT(name)); } CheckerContext ctx = {}; @@ -6603,15 +6555,6 @@ gb_internal bool check_proc_info(Checker *c, ProcInfo *pi, UntypedExprInfoMap *u add_untyped_expressions(&c->info, ctx.untyped); - rw_mutex_shared_lock(&ctx.decl->deps_mutex); - FOR_PTR_SET(dep, ctx.decl->deps) { - if (dep && dep->kind == Entity_Procedure && - (dep->flags & EntityFlag_ProcBodyChecked) == 0) { - check_procedure_later_from_entity(c, dep, NULL); - } - } - rw_mutex_shared_unlock(&ctx.decl->deps_mutex); - return true; } @@ -6619,38 +6562,37 @@ GB_STATIC_ASSERT(sizeof(isize) == sizeof(void *)); gb_internal bool consume_proc_info(Checker *c, ProcInfo *pi, UntypedExprInfoMap *untyped); -gb_internal void check_unchecked_bodies(Checker *c) { - // NOTE(2021-02-26, bill): Sanity checker - // This is a partial hack to make sure all procedure bodies have been checked - // even ones which should not exist, due to the multithreaded nature of the parser - // HACK TODO(2021-02-26, bill): Actually fix this race condition - +gb_internal void check_min_dep_bodies_were_checked(Checker *c) { GB_ASSERT(c->procs_to_check.count == 0); + global_after_checking_procedure_bodies = true; - UntypedExprInfoMap untyped = {}; - defer (map_destroy(&untyped)); - - // use the `procs_to_check` array - global_procedure_body_in_worker_queue = false; + if (any_errors()) { + // e.g. a body is not checked when its `where` clauses fail + return; + } for (Entity *e : c->info.entities) { - if (e->min_dep_count.load(std::memory_order_relaxed) > 0) { - check_procedure_later_from_entity(c, e, "check_unchecked_bodies"); + if (e->kind != Entity_Procedure || e->min_dep_count.load(std::memory_order_relaxed) == 0) { + continue; } - } - - if (!global_procedure_body_in_worker_queue) { - for_array(i, c->procs_to_check) { - ProcInfo *pi = c->procs_to_check[i]; - consume_proc_info(c, pi, &untyped); + Entity *original = e; + while ((e->flags & EntityFlag_Overridden) && e->aliased_of != nullptr) { + e = e->aliased_of; } - array_clear(&c->procs_to_check); - } else { - thread_pool_wait(); + if (e->Procedure.is_foreign || (e->flags & EntityFlag_ProcBodyChecked) != 0) { + continue; + } + Type *type = base_type(e->type); + if (type == t_invalid || (is_type_polymorphic(type) && !type->Proc.is_poly_specialized)) { + continue; + } + DeclInfo *decl = e->decl_info; + if (decl == nullptr || decl->proc_lit == nullptr || decl->proc_lit->ProcLit.body == nullptr) { + continue; + } + GB_PANIC("the body of '%.*s' :: %s was never checked (%s)", + LIT(original->token.string), type_to_string(e->type), token_pos_to_string(e->token.pos)); } - - global_procedure_body_in_worker_queue = false; - global_after_checking_procedure_bodies = true; } gb_internal void check_safety_all_procedures_for_unchecked(Checker *c) { @@ -7774,13 +7716,7 @@ gb_internal void check_parsed_files(Checker *c) { generate_minimum_dependency_set(c, c->info.entry_point); TIME_SECTION("check bodies have all been checked"); - check_unchecked_bodies(c); - - TIME_SECTION("check #soa types"); - check_merge_queues_into_arrays(c); - - TIME_SECTION("update minimum dependency set again"); - generate_minimum_dependency_set_internal(c, c->info.entry_point); + check_min_dep_bodies_were_checked(c); // NOTE(laytan): has to be ran after generate_minimum_dependency_set, // because that collects the test procedures. diff --git a/src/checker.hpp b/src/checker.hpp index 6c91c2b91..04cf1e41a 100644 --- a/src/checker.hpp +++ b/src/checker.hpp @@ -225,6 +225,7 @@ struct DeclInfo { Type * gen_proc_type; // Precalculated Entity * para_poly_original; + std::atomic gen_proc_info; // a specialization's body, queued for checking when it is first used bool is_using; bool foreign_require_results; From bdb03647390aa01412a09f6577cda798c28882df Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 22:26:08 +0100 Subject: [PATCH 09/20] Remove the missing procedures fallback from the backend: the checker adds the Objective-C runtime procedures the backend calls, local foreign procedures with a duplicate name are bound to the existing declaration, and local procedures passed to a `$` parameter are created up front --- src/check_builtin.cpp | 8 +++++ src/check_type.cpp | 1 + src/checker.cpp | 12 ++++++++ src/entity.cpp | 3 +- src/llvm_backend.cpp | 59 +++++++++++------------------------- src/llvm_backend.hpp | 2 -- src/llvm_backend_general.cpp | 16 ++-------- src/llvm_backend_stmt.cpp | 3 +- 8 files changed, 46 insertions(+), 58 deletions(-) diff --git a/src/check_builtin.cpp b/src/check_builtin.cpp index f453fb0aa..41e037eeb 100644 --- a/src/check_builtin.cpp +++ b/src/check_builtin.cpp @@ -686,6 +686,14 @@ gb_internal bool check_builtin_objc_procedure(CheckerContext *c, Operand *operan } else { try_to_add_package_dependency(c, "runtime", "_NSConcreteStackBlock"); } + for (isize i = 0; i < capture_arg_count; i++) { + Type *t = param_operands[i].type; + if (is_type_pointer(t) && is_type_objc_object(t)) { + try_to_add_package_dependency(c, "runtime", "_Block_object_assign"); + try_to_add_package_dependency(c, "runtime", "_Block_object_dispose"); + break; + } + } *operand = poly_op; operand->type = alloc_type_pointer(operand->type); diff --git a/src/check_type.cpp b/src/check_type.cpp index 98e562501..959ce39a9 100644 --- a/src/check_type.cpp +++ b/src/check_type.cpp @@ -3358,6 +3358,7 @@ gb_internal Type *check_get_params(CheckerContext *ctx, Scope *scope, Ast *_para Ast *expr = unparen_expr(op.expr); Entity *proc_entity = strip_entity_wrapping(expr); if (proc_entity) { + proc_entity->flags |= EntityFlag_PolyConstArg; poly_const = exact_value_procedure(proc_entity->identifier.load() ? proc_entity->identifier.load() : op.expr); valid = true; } else if (expr->kind == Ast_ProcLit) { diff --git a/src/checker.cpp b/src/checker.cpp index 0a4dec1fb..82e1dc154 100644 --- a/src/checker.cpp +++ b/src/checker.cpp @@ -3377,6 +3377,18 @@ gb_internal void generate_minimum_dependency_set(Checker *c, Entity *start) { str_lit("multi_pointer_slice_expr_error"), ); + FORCE_ADD_RUNTIME_ENTITIES(c->info.objc_class_implementations.count.load(std::memory_order_relaxed) > 0, + str_lit("objc_lookUpClass"), + str_lit("sel_registerName"), + str_lit("objc_allocateClassPair"), + str_lit("objc_registerClassPair"), + str_lit("class_addMethod"), + str_lit("class_addIvar"), + str_lit("class_getInstanceVariable"), + str_lit("ivar_getOffset"), + str_lit("object_getClass"), + ); + { // init min dep basic equal procs struct { BasicKind kind; char const *name; } const procs[] = { diff --git a/src/entity.cpp b/src/entity.cpp index 4a7800450..355bd1ca6 100644 --- a/src/entity.cpp +++ b/src/entity.cpp @@ -76,7 +76,8 @@ enum EntityFlag : u64 { EntityFlag_Init = 1ull<<31, EntityFlag_Subtype = 1ull<<32, EntityFlag_Fini = 1ull<<33, - + EntityFlag_PolyConstArg = 1ull<<34, // passed to a `$` parameter, so a local procedure may be called outside its parent + EntityFlag_CustomLinkName = 1ull<<40, EntityFlag_CustomLinkage_Internal = 1ull<<41, EntityFlag_CustomLinkage_Strong = 1ull<<42, diff --git a/src/llvm_backend.cpp b/src/llvm_backend.cpp index 900a762f9..2d25e4703 100644 --- a/src/llvm_backend.cpp +++ b/src/llvm_backend.cpp @@ -2441,12 +2441,13 @@ gb_internal void lb_create_global_procedures_and_types(lbGenerator *gen, Checker Scope * scope = e->scope; if ((scope->flags & ScopeFlag_File) == 0) { - continue; + if (e->kind != Entity_Procedure || (e->flags & EntityFlag_PolyConstArg) == 0) { + continue; + } + } else { + GB_ASSERT(scope->parent->flags & ScopeFlag_Pkg); } - Scope *package_scope = scope->parent; - GB_ASSERT(package_scope->flags & ScopeFlag_Pkg); - switch (e->kind) { case Entity_Variable: // NOTE(bill): Handled above as it requires a specific load order @@ -3354,51 +3355,27 @@ gb_internal void lb_generate_procedures(lbGenerator *gen, bool do_threading) { lb_exit_if_worker_failed(); } -gb_internal WORKER_TASK_PROC(lb_generate_missing_procedures_to_check_worker_proc) { +gb_internal WORKER_TASK_PROC(lb_generate_queued_procedures_worker_proc) { lbModule *m = cast(lbModule *)data; - for (Entity *e = nullptr; mpsc_dequeue(&m->missing_procedures_to_check, &e); /**/) { - lbProcedure *p = lb_create_procedure(m, e, false); - if (!p->is_done.load(std::memory_order_relaxed)) { - debugf("Generate missing procedure: %.*s module %p\n", LIT(p->name), m); - } - mpsc_enqueue(&m->procedures_to_generate, p); - } for (lbProcedure *p = nullptr; mpsc_dequeue(&m->procedures_to_generate, &p); /**/) { lb_generate_procedure(m, p); } return 0; } -gb_internal void lb_generate_missing_procedures(lbGenerator *gen, bool do_threading) { - isize retry_count = 0; -retry:; - if (do_threading) { - for (auto const &entry : gen->modules) { - lbModule *m = entry.value; - // NOTE(bill): procedures may be added during generation - thread_pool_add_task(lb_generate_missing_procedures_to_check_worker_proc, m); - } - thread_pool_wait(); - } else { - for (auto const &entry : gen->modules) { - lbModule *m = entry.value; - // NOTE(bill): procedures may be added during generation - lb_generate_missing_procedures_to_check_worker_proc(m); - } - } - +gb_internal void lb_generate_queued_procedures(lbGenerator *gen, bool do_threading) { for (auto const &entry : gen->modules) { lbModule *m = entry.value; - if (m->missing_procedures_to_check.count != 0 || m->procedures_to_generate.count != 0) { - if (retry_count > gen->modules.count) { - GB_ASSERT(m->missing_procedures_to_check.count == 0 && m->procedures_to_generate.count == 0); - } - - retry_count += 1; - goto retry; + if (do_threading) { + thread_pool_add_task(lb_generate_queued_procedures_worker_proc, m); + } else { + lb_generate_queued_procedures_worker_proc(m); } - GB_ASSERT(m->missing_procedures_to_check.count == 0); - GB_ASSERT(m->procedures_to_generate.count == 0); + } + thread_pool_wait(); + + for (auto const &entry : gen->modules) { + GB_ASSERT(entry.value->procedures_to_generate.count == 0); } } @@ -4236,8 +4213,8 @@ gb_internal bool lb_generate_code(lbGenerator *gen) { lb_create_main_procedure(default_module, gen->startup_runtime, gen->cleanup_runtime); } - TIME_SECTION("LLVM Procedure Generation (missing)"); - lb_generate_missing_procedures(gen, do_threading); + TIME_SECTION("LLVM Procedure Generation (queued)"); + lb_generate_queued_procedures(gen, do_threading); if (gen->objc_names) { TIME_SECTION("Finalize objc names"); diff --git a/src/llvm_backend.hpp b/src/llvm_backend.hpp index 824fe2f88..7a3b12575 100644 --- a/src/llvm_backend.hpp +++ b/src/llvm_backend.hpp @@ -153,8 +153,6 @@ struct lbModule { StringMap procedures; PtrMap procedure_values; - MPSCQueue missing_procedures_to_check; - StringMap const_strings; String16Map const_string16s; diff --git a/src/llvm_backend_general.cpp b/src/llvm_backend_general.cpp index a2b2a44b9..9893debbf 100644 --- a/src/llvm_backend_general.cpp +++ b/src/llvm_backend_general.cpp @@ -165,7 +165,6 @@ gb_internal WORKER_TASK_PROC(lb_init_module_worker_proc) { array_init(&m->global_procedures_to_create, a, 0, 1024); array_init(&m->global_types_to_create, a, 0, 1024); array_init(&m->global_variables, a); - mpsc_init(&m->missing_procedures_to_check, a); map_init(&m->debug_values); string_map_init(&m->objc_classes); @@ -4086,25 +4085,16 @@ gb_internal lbValue lb_find_procedure_value_from_entity(lbModule *m, Entity *e) // NOTE(bill): Until the modules are generated in parallel, a procedure may be referenced before it is created // (e.g. an @(init) procedure by the startup procedure), but after that it was missed by the frontend - bool missing = gen->modules_in_parallel; - if (!ignore_body) { - if (missing) { - debugf("Missing Procedure (lb_find_procedure_value_from_entity): %.*s module %p\n", LIT(e->token.string), m); - } + GB_ASSERT_MSG(!gen->modules_in_parallel, "missing procedure '%.*s' (%s)", LIT(e->token.string), token_pos_to_string(e->token.pos)); mpsc_enqueue(&m->procedures_to_generate, proc); } else { rw_mutex_shared_lock(&other_module->values_mutex); auto *found = map_get(&other_module->values, e); rw_mutex_shared_unlock(&other_module->values_mutex); if (found == nullptr) { - if (missing) { - debugf("Missing Procedure (lb_find_procedure_value_from_entity): %.*s module %p\n", LIT(e->token.string), other_module); - // another module's context may only be used by the thread generating that module - mpsc_enqueue(&other_module->missing_procedures_to_check, e); - } else { - mpsc_enqueue(&other_module->procedures_to_generate, lb_create_procedure(other_module, e, false)); - } + GB_ASSERT_MSG(!gen->modules_in_parallel, "missing procedure '%.*s' (%s)", LIT(e->token.string), token_pos_to_string(e->token.pos)); + mpsc_enqueue(&other_module->procedures_to_generate, lb_create_procedure(other_module, e, false)); } } diff --git a/src/llvm_backend_stmt.cpp b/src/llvm_backend_stmt.cpp index d35b48835..eb51e6370 100644 --- a/src/llvm_backend_stmt.cpp +++ b/src/llvm_backend_stmt.cpp @@ -234,7 +234,8 @@ gb_internal void lb_build_constant_value_decl(lbProcedure *p, AstValueDecl *vd) lbValue *prev_value = string_map_get(&p->module->members, name); if (prev_value != nullptr) { // NOTE(bill): Don't do mutliple declarations in the IR - return; + lb_add_entity(p->module, e, *prev_value); + continue; } e->Procedure.link_name = name; From 733c3fdea73861109d6e7036dd475fb871006a66 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 22:31:40 +0100 Subject: [PATCH 10/20] Add to the tests the equal procedure dependencies test --- tests/issues/run.bat | 2 +- tests/issues/run.sh | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/issues/run.bat b/tests/issues/run.bat index 75d94c400..9c12b346d 100644 --- a/tests/issues/run.bat +++ b/tests/issues/run.bat @@ -99,7 +99,7 @@ clang -c ..\test_issue_sysv_abi.c -o test_issue_sysv_abi_c.o || exit /b ..\..\..\odin test ..\test_issue_fast_isel_lowering.odin %COMMON% || exit /b ..\..\..\odin test ..\test_issue_fast_isel_lowering.odin %COMMON% -o:none || exit /b ..\..\..\odin test ..\test_issue_equal_proc_dependencies.odin %COMMON% || exit /b -..\..\..\odin test ..\test_issue_equal_proc_dependencies.odin %COMMON% -build-mode:obj -show-debug-messages 2>&1 | find /i /c "missing procedure" | findstr /x "0" || exit /b +..\..\..\odin test ..\test_issue_equal_proc_dependencies.odin %COMMON% -build-mode:obj 2>&1 | find /i /c "Assertion Failure" | findstr /x "0" || exit /b @echo off diff --git a/tests/issues/run.sh b/tests/issues/run.sh index e14677895..c0684abeb 100755 --- a/tests/issues/run.sh +++ b/tests/issues/run.sh @@ -125,7 +125,7 @@ $ODIN test ../test_issue_omitted_field_union.odin $COMMON $ODIN test ../test_issue_fast_isel_lowering.odin $COMMON $ODIN test ../test_issue_fast_isel_lowering.odin $COMMON -o:none $ODIN test ../test_issue_equal_proc_dependencies.odin $COMMON -if [[ $($ODIN test ../test_issue_equal_proc_dependencies.odin $COMMON -build-mode:obj -show-debug-messages 2>&1 | grep -ci "missing procedure") -eq 0 ]]; then +if [[ $($ODIN test ../test_issue_equal_proc_dependencies.odin $COMMON -build-mode:obj 2>&1 | grep -ci "Assertion Failure") -eq 0 ]]; then echo "SUCCESSFUL 1/1" else echo "SUCCESSFUL 0/1" From 5a4b877beb336439044d3daa8903ad18b47e3f8b Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 22:33:08 +0100 Subject: [PATCH 11/20] Only match the missing procedure assertion in the equal procedure dependencies issue test --- tests/issues/run.bat | 2 +- tests/issues/run.sh | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/issues/run.bat b/tests/issues/run.bat index 9c12b346d..5c459a910 100644 --- a/tests/issues/run.bat +++ b/tests/issues/run.bat @@ -99,7 +99,7 @@ clang -c ..\test_issue_sysv_abi.c -o test_issue_sysv_abi_c.o || exit /b ..\..\..\odin test ..\test_issue_fast_isel_lowering.odin %COMMON% || exit /b ..\..\..\odin test ..\test_issue_fast_isel_lowering.odin %COMMON% -o:none || exit /b ..\..\..\odin test ..\test_issue_equal_proc_dependencies.odin %COMMON% || exit /b -..\..\..\odin test ..\test_issue_equal_proc_dependencies.odin %COMMON% -build-mode:obj 2>&1 | find /i /c "Assertion Failure" | findstr /x "0" || exit /b +..\..\..\odin test ..\test_issue_equal_proc_dependencies.odin %COMMON% -build-mode:obj 2>&1 | find /i /c "missing procedure" | findstr /x "0" || exit /b @echo off diff --git a/tests/issues/run.sh b/tests/issues/run.sh index c0684abeb..c4596914c 100755 --- a/tests/issues/run.sh +++ b/tests/issues/run.sh @@ -125,7 +125,7 @@ $ODIN test ../test_issue_omitted_field_union.odin $COMMON $ODIN test ../test_issue_fast_isel_lowering.odin $COMMON $ODIN test ../test_issue_fast_isel_lowering.odin $COMMON -o:none $ODIN test ../test_issue_equal_proc_dependencies.odin $COMMON -if [[ $($ODIN test ../test_issue_equal_proc_dependencies.odin $COMMON -build-mode:obj 2>&1 | grep -ci "Assertion Failure") -eq 0 ]]; then +if [[ $($ODIN test ../test_issue_equal_proc_dependencies.odin $COMMON -build-mode:obj 2>&1 | grep -ci "missing procedure") -eq 0 ]]; then echo "SUCCESSFUL 1/1" else echo "SUCCESSFUL 0/1" From 1396a36bd31cc81e90d3e5cd8cab0ca497455db5 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sat, 3 Oct 2026 23:25:17 +0100 Subject: [PATCH 12/20] Prefer `/OPT:NOICF` when using `-debug` on Windows to improve linking performance --- src/linker.cpp | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/linker.cpp b/src/linker.cpp index 519480826..c45aa4bd8 100644 --- a/src/linker.cpp +++ b/src/linker.cpp @@ -384,6 +384,10 @@ try_cross_linking:; if (build_context.ODIN_DEBUG) { link_settings = gb_string_append_fmt(link_settings, " /DEBUG"); + if (build_context.build_mode != BuildMode_StaticLibrary) { + // NOTE(bill): `/opt:ref` would also fold identical functions, which is slow and confuses the debugger + link_settings = gb_string_append_fmt(link_settings, " /OPT:NOICF"); + } } gbString object_files = gb_string_make(heap_allocator(), ""); From 55eecd010ab65b60666bdc06750e16425f6a6472 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 09:58:15 +0100 Subject: [PATCH 13/20] In a ternary if expr, large aggregate need to be selected through memory --- src/llvm_backend_expr.cpp | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/src/llvm_backend_expr.cpp b/src/llvm_backend_expr.cpp index a1e99e7b6..10ad9a367 100644 --- a/src/llvm_backend_expr.cpp +++ b/src/llvm_backend_expr.cpp @@ -4710,6 +4710,28 @@ gb_internal lbValue lb_build_expr_internal(lbProcedure *p, Ast *expr) { case_ast_node(te, TernaryIfExpr, expr); GB_ASSERT(te->y != nullptr); Type *type = default_type(type_of_expr(expr)); + LLVMTypeKind type_kind = LLVMGetTypeKind(lb_type(p->module, type)); + if ((type_kind == LLVMStructTypeKind || type_kind == LLVMArrayTypeKind) && type_size_of(type) > 64) { + // NOTE(bill): A large aggregate needs to be selected through memory + // as instruction selection splits a `phi` or `select` of it per field + lbAddr res = lb_add_local_generated(p, type, false); + + lbBlock *then = lb_create_block(p, "if.then"); + lbBlock *done = lb_create_block(p, "if.done"); + lbBlock *else_ = lb_create_block(p, "if.else"); + + lb_build_cond(p, te->cond, then, else_); + lb_start_block(p, then); + lb_addr_store(p, res, lb_emit_conv(p, lb_build_expr(p, te->x), type)); + lb_emit_jump(p, done); + + lb_start_block(p, else_); + lb_addr_store(p, res, lb_emit_conv(p, lb_build_expr(p, te->y), type)); + lb_emit_jump(p, done); + + lb_start_block(p, done); + return lb_addr_load(p, res); + } if (lb_is_expr_trivial(te->x) && lb_is_expr_trivial(te->y)) { lbValue cond = lb_build_expr(p, te->cond); lbValue x = lb_emit_conv(p, lb_build_expr(p, te->x), type); From 6fbc112088454cec05116a68b5c7dc9189751dd8 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 10:05:19 +0100 Subject: [PATCH 14/20] Lower more instructions for the fast instruction selector in LLVM --- src/llvm_backend.cpp | 100 ++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 98 insertions(+), 2 deletions(-) diff --git a/src/llvm_backend.cpp b/src/llvm_backend.cpp index 2d25e4703..89c9139d8 100644 --- a/src/llvm_backend.cpp +++ b/src/llvm_backend.cpp @@ -2820,6 +2820,98 @@ gb_internal bool lb_scalarize_aggregate_store(lbFastIselLowering *s, LLVMValueRe return true; } +// NOTE(bill): If a a constant has too many fields to store one at a time is set or copied from a constant instead +gb_internal void lb_lower_large_constant_store(lbFastIselLowering *s, LLVMValueRef store) { + lbModule *m = s->m; + LLVMValueRef value = LLVMGetOperand(store, 0); + LLVMValueRef ptr = LLVMGetOperand(store, 1); + LLVMTypeRef type = LLVMTypeOf(value); + + LLVMTargetDataRef data_layout = LLVMGetModuleDataLayout(m->mod); + unsigned alignment = gb_max(LLVMGetAlignment(store), 1u); + if (!LLVMIsAConstant(value) || lb_const_has_misaligned_pointer(data_layout, value, 0, alignment)) { + return; + } + + LLVMPositionBuilderBefore(s->builder, store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); + + LLVMValueRef size = LLVMConstInt(LLVMInt64TypeInContext(m->ctx), LLVMStoreSizeOfType(data_layout, type), false); + if (LLVMIsNull(value)) { + LLVMBuildMemSet(s->builder, ptr, LLVMConstInt(LLVMInt8TypeInContext(m->ctx), 0, false), size, alignment); + LLVMInstructionEraseFromParent(store); + return; + } + + LLVMValueRef global = LLVMAddGlobal(m->mod, type, ""); + LLVMSetInitializer(global, value); + LLVMSetGlobalConstant(global, true); + LLVMSetAlignment(global, alignment); + + LLVMSetLinkage(global, LLVMPrivateLinkage); + LLVMSetUnnamedAddress(global, LLVMGlobalUnnamedAddr); + + LLVMBuildMemCpy(s->builder, ptr, alignment, global, alignment, size); + + LLVMInstructionEraseFromParent(store); +} + +// NOTE(bill): The fast instruction selector in LLVM cannot select a call to `memmove` nor to the floating point `minnum` and `maxnum`. +// To improve things we can them through a function of the module which calls them +gb_internal void lb_redirect_unselectable_call(lbFastIselLowering *s, LLVMValueRef call) { + lbModule *m = s->m; + + LLVMValueRef callee = LLVMGetCalledValue(call); + if (!LLVMIsAFunction(callee)) { + return; + } + size_t name_len = 0; + char const *name_text = LLVMGetValueName2(callee, &name_len); + String name = make_string(cast(u8 const *)name_text, name_len); + + LLVMTypeRef fn_type = LLVMGlobalGetValueType(callee); + if (LLVMGetCalledFunctionType(call) != fn_type || LLVMIsFunctionVarArg(fn_type)) { + return; + } + unsigned param_count = LLVMCountParamTypes(fn_type); + LLVMTypeKind return_kind = LLVMGetTypeKind(LLVMGetReturnType(fn_type)); + bool is_memmove = name == "memmove" && param_count == 3 && return_kind == LLVMPointerTypeKind; + bool is_minmax = (string_starts_with(name, str_lit("llvm.minnum.")) || + string_starts_with(name, str_lit("llvm.maxnum."))) && + (return_kind == LLVMFloatTypeKind || + return_kind == LLVMDoubleTypeKind); + if (!is_memmove && !is_minmax) { + return; + } + + gbString wrapper_name = gb_string_make(heap_allocator(), "__$fast_isel$"); + wrapper_name = gb_string_append_length(wrapper_name, name.text, name.len); + defer (gb_string_free(wrapper_name)); + + LLVMValueRef wrapper = LLVMGetNamedFunction(m->mod, wrapper_name); + defer (LLVMSetOperand(call, cast(unsigned)LLVMGetNumOperands(call) - 1, wrapper)); + + if (wrapper != nullptr) { + return; + } + + LLVMCallConv cc = cast(LLVMCallConv)LLVMGetFunctionCallConv(callee); + wrapper = LLVMAddFunction(m->mod, wrapper_name, fn_type); + LLVMSetLinkage(wrapper, LLVMInternalLinkage); + LLVMSetFunctionCallConv(wrapper, cc); + lb_add_attribute_to_proc(m, wrapper, "nounwind"); + + LLVMValueRef params[3] = {}; + LLVMGetParams(wrapper, params); + + LLVMPositionBuilderAtEnd(s->builder, LLVMAppendBasicBlockInContext(m->ctx, wrapper, "")); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + LLVMValueRef inner = LLVMBuildCall2(s->builder, fn_type, callee, params, param_count, ""); + LLVMSetInstructionCallConv(inner, cc); + LLVMBuildRet(s->builder, inner); +} + gb_internal bool lb_is_aggregate_type(LLVMTypeRef type) { LLVMTypeKind kind = LLVMGetTypeKind(type); return kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind; @@ -2986,7 +3078,8 @@ gb_internal void lb_lower_for_fast_isel(lbModule *m) { auto work = array_make(heap_allocator(), 0, 64); defer (array_free(&work)); - for (LLVMValueRef fn = LLVMGetFirstFunction(m->mod); fn != nullptr; fn = LLVMGetNextFunction(fn)) { + LLVMValueRef last_fn = LLVMGetLastFunction(m->mod); + for (LLVMValueRef fn = LLVMGetFirstFunction(m->mod); fn != nullptr; fn = fn == last_fn ? nullptr : LLVMGetNextFunction(fn)) { array_clear(&work); array_clear(&s.phis); @@ -2998,7 +3091,9 @@ gb_internal void lb_lower_for_fast_isel(lbModule *m) { } } for (LLVMValueRef store : work) { - lb_scalarize_aggregate_store(&s, store); + if (!lb_scalarize_aggregate_store(&s, store)) { + lb_lower_large_constant_store(&s, store); + } } // NOTE(bill): A field read out of an aggregate value is read where that aggregate came from @@ -3023,6 +3118,7 @@ gb_internal void lb_lower_for_fast_isel(lbModule *m) { } if (LLVMIsACallInst(i)) { lb_truncate_bool_arguments(&s, i); + lb_redirect_unselectable_call(&s, i); continue; } if (LLVMIsASwitchInst(i)) { From ddcf115f6f891aa50046577c2922340aa84744ed Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 10:12:35 +0100 Subject: [PATCH 15/20] Implement the same large aggregate improvement for `or_else` --- src/llvm_backend_expr.cpp | 3 +-- src/llvm_backend_utility.cpp | 24 ++++++++++++++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/src/llvm_backend_expr.cpp b/src/llvm_backend_expr.cpp index 10ad9a367..de9aee30b 100644 --- a/src/llvm_backend_expr.cpp +++ b/src/llvm_backend_expr.cpp @@ -4710,8 +4710,7 @@ gb_internal lbValue lb_build_expr_internal(lbProcedure *p, Ast *expr) { case_ast_node(te, TernaryIfExpr, expr); GB_ASSERT(te->y != nullptr); Type *type = default_type(type_of_expr(expr)); - LLVMTypeKind type_kind = LLVMGetTypeKind(lb_type(p->module, type)); - if ((type_kind == LLVMStructTypeKind || type_kind == LLVMArrayTypeKind) && type_size_of(type) > 64) { + if (lb_is_type_large_aggregate(p->module, type)) { // NOTE(bill): A large aggregate needs to be selected through memory // as instruction selection splits a `phi` or `select` of it per field lbAddr res = lb_add_local_generated(p, type, false); diff --git a/src/llvm_backend_utility.cpp b/src/llvm_backend_utility.cpp index aa33073c5..a4385c883 100644 --- a/src/llvm_backend_utility.cpp +++ b/src/llvm_backend_utility.cpp @@ -489,6 +489,11 @@ gb_internal bool lb_is_type_trivial(Type *type) { return false; } +gb_internal bool lb_is_type_large_aggregate(lbModule *m, Type *type) { + LLVMTypeKind kind = LLVMGetTypeKind(lb_type(m, type)); + return (kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind) && type_size_of(type) > 64; +} + gb_internal bool lb_is_expr_trivial(Ast *e) { Type *type = default_type(type_of_expr(e)); if (lb_is_type_trivial(type)) { @@ -556,6 +561,25 @@ gb_internal lbValue lb_emit_or_else(lbProcedure *p, Ast *arg, Ast *else_expr, Ty } return {}; } else { + if (lb_is_type_large_aggregate(p->module, type)) { + lbAddr res = lb_add_local_generated(p, type, false); + + lbBlock *then = lb_create_block(p, "or_else.then"); + lbBlock *done = lb_create_block(p, "or_else.done"); + lbBlock *else_ = lb_create_block(p, "or_else.else"); + + lb_emit_if(p, lb_emit_try_has_value(p, rhs), then, else_); + lb_start_block(p, then); + lb_addr_store(p, res, lb_emit_conv(p, lhs, type)); + lb_emit_jump(p, done); + + lb_start_block(p, else_); + lb_addr_store(p, res, lb_emit_conv(p, lb_build_expr(p, else_expr), type)); + lb_emit_jump(p, done); + + lb_start_block(p, done); + return lb_addr_load(p, res); + } if (lb_is_type_trivial(type) && lb_is_expr_trivial(else_expr)) { lbValue has_value = lb_emit_try_has_value(p, rhs); lbValue then_val = lb_emit_conv(p, lhs, type); From ccf1ae4fe9d3988f3cdf6285dc9441bfd342f325 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 10:16:50 +0100 Subject: [PATCH 16/20] Move the lowering for LLVM's fast instruction selector to `llvm_backend_opt.cpp` --- src/llvm_backend.cpp | 621 ----------------------------------- src/llvm_backend.hpp | 3 + src/llvm_backend_general.cpp | 2 +- src/llvm_backend_opt.cpp | 619 ++++++++++++++++++++++++++++++++++ 4 files changed, 623 insertions(+), 622 deletions(-) diff --git a/src/llvm_backend.cpp b/src/llvm_backend.cpp index 89c9139d8..134b776aa 100644 --- a/src/llvm_backend.cpp +++ b/src/llvm_backend.cpp @@ -2546,627 +2546,6 @@ gb_internal bool lb_is_module_empty(lbModule *m) { return true; } -// NOTE(bill, 2026-10-03) -// -// LLVM's fast instruction selector cannot select a first class aggregate `load`, `store`, `insertvalue`, or `select`, -// and hands the rest of the block to SelectionDAG. -// SROA leaves many behind so it stores/loads the fields instead -enum { - LB_SCALARIZE_MAX_LEAVES = 32, - LB_SCALARIZE_MAX_DEPTH = 8, -}; - -struct lbAggregateLeaf { - unsigned path[LB_SCALARIZE_MAX_DEPTH]; - unsigned depth; -}; - -struct lbScalarizedPhi { - LLVMValueRef aggregate; - unsigned path[LB_SCALARIZE_MAX_DEPTH]; - unsigned depth; - LLVMValueRef field; - bool ok; -}; - -struct lbFastIselLowering { - lbModule * m; - LLVMBuilderRef builder; - LLVMValueRef store; - Array phis; -}; - -gb_internal i64 lb_aggregate_path_offset(LLVMTypeRef type, unsigned const *path, unsigned depth, LLVMTypeRef *leaf_type_) { - i64 offset = 0; - for (unsigned d = 0; d < depth; d++) { - if (LLVMGetTypeKind(type) == LLVMStructTypeKind) { - bool is_packed = LLVMIsPackedStruct(type); - i64 field_offset = 0; - for (unsigned i = 0; i <= path[d]; i++) { - LLVMTypeRef field = LLVMStructGetTypeAtIndex(type, i); - if (!is_packed) { - field_offset = llvm_align_formula(field_offset, lb_alignof(field)); - } - if (i == path[d]) { - type = field; - break; - } - field_offset += lb_sizeof(field); - } - offset += field_offset; - } else { - type = OdinLLVMGetArrayElementType(type); - offset += cast(i64)path[d] * lb_sizeof(type); - } - } - if (leaf_type_) *leaf_type_ = type; - return offset; -} - -gb_internal bool lb_aggregate_leaves(LLVMTypeRef type, unsigned *path, unsigned depth, lbAggregateLeaf *leaves, isize *leaf_count) { - LLVMTypeKind kind = LLVMGetTypeKind(type); - if (kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind) { - if (depth >= LB_SCALARIZE_MAX_DEPTH) { - return false; - } - if (kind == LLVMStructTypeKind && LLVMIsOpaqueStruct(type)) { - return false; - } - unsigned count = kind == LLVMStructTypeKind ? LLVMCountStructElementTypes(type) : cast(unsigned)LLVMGetArrayLength(type); - for (unsigned i = 0; i < count; i++) { - path[depth] = i; - LLVMTypeRef elem = kind == LLVMStructTypeKind ? LLVMStructGetTypeAtIndex(type, i) : OdinLLVMGetArrayElementType(type); - if (!lb_aggregate_leaves(elem, path, depth+1, leaves, leaf_count)) { - return false; - } - } - return true; - } - - if (*leaf_count >= LB_SCALARIZE_MAX_LEAVES) { - return false; - } - - lbAggregateLeaf *leaf = &leaves[(*leaf_count)++]; - gb_memmove(leaf->path, path, depth*gb_size_of(unsigned)); - leaf->depth = depth; - return true; -} - -gb_internal unsigned lb_alignment_at_offset(unsigned alignment, i64 offset) { - unsigned a = gb_max(alignment, 1u); - while (offset % a != 0) { - a >>= 1; - } - return a; -} - -gb_internal LLVMValueRef lb_aggregate_field_gep(lbModule *m, LLVMBuilderRef b, LLVMTypeRef type, LLVMValueRef ptr, unsigned const *path, unsigned depth) { - LLVMValueRef indices[LB_SCALARIZE_MAX_DEPTH+1] = {}; - LLVMTypeRef i32 = LLVMInt32TypeInContext(m->ctx); - indices[0] = LLVMConstInt(i32, 0, false); - for (unsigned d = 0; d < depth; d++) { - indices[d+1] = LLVMConstInt(i32, path[d], false); - } - return LLVMBuildInBoundsGEP2(b, type, ptr, indices, depth+1, ""); -} - -// returns false if the field cannot be reached without an aggregate value, and sets `*field_` to nullptr when it is undefined -gb_internal bool lb_aggregate_field_value(lbFastIselLowering *s, LLVMValueRef v, unsigned const *path, unsigned depth, LLVMValueRef *field_) { - if (depth == 0) { - *field_ = (LLVMIsUndef(v) || LLVMIsPoison(v)) ? nullptr : v; - return true; - } - if (LLVMIsUndef(v) || LLVMIsPoison(v)) { - *field_ = nullptr; - return true; - } - - if (LLVMIsAConstant(v)) { - LLVMValueRef elem = LLVMGetAggregateElement(v, path[0]); - if (elem != nullptr) { - return lb_aggregate_field_value(s, elem, path+1, depth-1, field_); - } - } else if (LLVMIsAInsertValueInst(v)) { - unsigned n = LLVMGetNumIndices(v); - unsigned const *indices = LLVMGetIndices(v); - - unsigned k = 0; - while (k < n && k < depth && indices[k] == path[k]) { - k += 1; - } - - if (k == n) { - return lb_aggregate_field_value(s, LLVMGetOperand(v, 1), path+n, depth-n, field_); - } else if (k < depth) { - return lb_aggregate_field_value(s, LLVMGetOperand(v, 0), path, depth, field_); - } - return false; - } else if (LLVMIsAExtractValueInst(v)) { - unsigned n = LLVMGetNumIndices(v); - if (n + depth <= LB_SCALARIZE_MAX_DEPTH) { - unsigned full[LB_SCALARIZE_MAX_DEPTH]; - gb_memmove(full, LLVMGetIndices(v), n*gb_size_of(unsigned)); - gb_memmove(full+n, path, depth*gb_size_of(unsigned)); - return lb_aggregate_field_value(s, LLVMGetOperand(v, 0), full, n+depth, field_); - } - return false; - } else if (LLVMIsASelectInst(v)) { - LLVMValueRef x = nullptr; - LLVMValueRef y = nullptr; - if (!lb_aggregate_field_value(s, LLVMGetOperand(v, 1), path, depth, &x) || - !lb_aggregate_field_value(s, LLVMGetOperand(v, 2), path, depth, &y)) { - return false; - } - if (x == nullptr || y == nullptr) { - *field_ = x ? x : y; - return true; - } - LLVMPositionBuilderBefore(s->builder, s->store); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(s->store)); - *field_ = LLVMBuildSelect(s->builder, LLVMGetOperand(v, 0), x, y, ""); - return true; - } else if (LLVMIsAPHINode(v)) { - for (lbScalarizedPhi const &e : s->phis) { - if (e.aggregate == v && - e.depth == depth && - gb_memcompare(e.path, path, depth*gb_size_of(unsigned)) == 0) { - *field_ = e.field; - return e.ok; - } - } - - LLVMTypeRef field_type = nullptr; - lb_aggregate_path_offset(LLVMTypeOf(v), path, depth, &field_type); - - LLVMPositionBuilderBefore(s->builder, v); - LLVMSetCurrentDebugLocation2(s->builder, nullptr); - - - // NOTE(bill): cached before its incoming values, as a loop reaches it again - isize index = s->phis.count; - lbScalarizedPhi entry = {}; - entry.aggregate = v; - gb_memmove(entry.path, path, depth*gb_size_of(unsigned)); - - LLVMValueRef phi = LLVMBuildPhi(s->builder, field_type, ""); - entry.depth = depth; - entry.field = phi; - entry.ok = true; - array_add(&s->phis, entry); - - LLVMValueRef saved_store = s->store; - bool ok = true; - unsigned incoming_count = LLVMCountIncoming(v); - for (unsigned k = 0; k < incoming_count; k++) { - LLVMBasicBlockRef block = LLVMGetIncomingBlock(v, k); - s->store = LLVMGetBasicBlockTerminator(block); - LLVMValueRef field = nullptr; - if (!lb_aggregate_field_value(s, LLVMGetIncomingValue(v, k), path, depth, &field)) { - ok = false; - field = nullptr; - } - if (field == nullptr) { - field = LLVMGetUndef(field_type); - } - LLVMAddIncoming(phi, &field, &block, 1); - } - s->store = saved_store; - s->phis[index].ok = ok; - *field_ = phi; - return ok; - } else if (LLVMIsALoadInst(v) && !LLVMGetVolatile(v) && LLVMGetOrdering(v) == LLVMAtomicOrderingNotAtomic) { - // NOTE(bill): Read the field where the aggregate was read, as the memory may change before the store - LLVMTypeRef type = LLVMTypeOf(v); - LLVMTypeRef field_type = nullptr; - i64 offset = lb_aggregate_path_offset(type, path, depth, &field_type); - - LLVMPositionBuilderBefore(s->builder, LLVMGetNextInstruction(v)); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(v)); - - LLVMValueRef ptr = lb_aggregate_field_gep(s->m, s->builder, type, LLVMGetOperand(v, 0), path, depth); - LLVMValueRef field = LLVMBuildLoad2(s->builder, field_type, ptr, ""); - LLVMSetAlignment(field, lb_alignment_at_offset(LLVMGetAlignment(v), offset)); - - *field_ = field; - return true; - } - - if (depth == 1) { - LLVMPositionBuilderBefore(s->builder, s->store); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(s->store)); - *field_ = LLVMBuildExtractValue(s->builder, v, path[0], ""); - return true; - } - return false; -} - -gb_internal bool lb_scalarize_aggregate_store(lbFastIselLowering *s, LLVMValueRef store) { - LLVMValueRef value = LLVMGetOperand(store, 0); - LLVMValueRef ptr = LLVMGetOperand(store, 1); - LLVMTypeRef type = LLVMTypeOf(value); - - lbAggregateLeaf leaves[LB_SCALARIZE_MAX_LEAVES] = {}; - isize leaf_count = 0; - unsigned path[LB_SCALARIZE_MAX_DEPTH] = {}; - - if (!lb_aggregate_leaves(type, path, 0, leaves, &leaf_count)) { - return false; - } - s->store = store; - - LLVMValueRef fields[LB_SCALARIZE_MAX_LEAVES] = {}; - for (isize i = 0; i < leaf_count; i++) { - if (!lb_aggregate_field_value(s, value, leaves[i].path, leaves[i].depth, &fields[i])) { - return false; - } - } - - unsigned alignment = LLVMGetAlignment(store); - - LLVMPositionBuilderBefore(s->builder, store); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); - - for (isize i = 0; i < leaf_count; i++) { - if (fields[i] == nullptr) { - continue; - } - i64 offset = lb_aggregate_path_offset(type, leaves[i].path, leaves[i].depth, nullptr); - LLVMValueRef field_ptr = lb_aggregate_field_gep(s->m, s->builder, type, ptr, leaves[i].path, leaves[i].depth); - LLVMValueRef field_store = LLVMBuildStore(s->builder, fields[i], field_ptr); - LLVMSetAlignment(field_store, lb_alignment_at_offset(alignment, offset)); - } - LLVMInstructionEraseFromParent(store); - return true; -} - -// NOTE(bill): If a a constant has too many fields to store one at a time is set or copied from a constant instead -gb_internal void lb_lower_large_constant_store(lbFastIselLowering *s, LLVMValueRef store) { - lbModule *m = s->m; - LLVMValueRef value = LLVMGetOperand(store, 0); - LLVMValueRef ptr = LLVMGetOperand(store, 1); - LLVMTypeRef type = LLVMTypeOf(value); - - LLVMTargetDataRef data_layout = LLVMGetModuleDataLayout(m->mod); - unsigned alignment = gb_max(LLVMGetAlignment(store), 1u); - if (!LLVMIsAConstant(value) || lb_const_has_misaligned_pointer(data_layout, value, 0, alignment)) { - return; - } - - LLVMPositionBuilderBefore(s->builder, store); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); - - LLVMValueRef size = LLVMConstInt(LLVMInt64TypeInContext(m->ctx), LLVMStoreSizeOfType(data_layout, type), false); - if (LLVMIsNull(value)) { - LLVMBuildMemSet(s->builder, ptr, LLVMConstInt(LLVMInt8TypeInContext(m->ctx), 0, false), size, alignment); - LLVMInstructionEraseFromParent(store); - return; - } - - LLVMValueRef global = LLVMAddGlobal(m->mod, type, ""); - LLVMSetInitializer(global, value); - LLVMSetGlobalConstant(global, true); - LLVMSetAlignment(global, alignment); - - LLVMSetLinkage(global, LLVMPrivateLinkage); - LLVMSetUnnamedAddress(global, LLVMGlobalUnnamedAddr); - - LLVMBuildMemCpy(s->builder, ptr, alignment, global, alignment, size); - - LLVMInstructionEraseFromParent(store); -} - -// NOTE(bill): The fast instruction selector in LLVM cannot select a call to `memmove` nor to the floating point `minnum` and `maxnum`. -// To improve things we can them through a function of the module which calls them -gb_internal void lb_redirect_unselectable_call(lbFastIselLowering *s, LLVMValueRef call) { - lbModule *m = s->m; - - LLVMValueRef callee = LLVMGetCalledValue(call); - if (!LLVMIsAFunction(callee)) { - return; - } - size_t name_len = 0; - char const *name_text = LLVMGetValueName2(callee, &name_len); - String name = make_string(cast(u8 const *)name_text, name_len); - - LLVMTypeRef fn_type = LLVMGlobalGetValueType(callee); - if (LLVMGetCalledFunctionType(call) != fn_type || LLVMIsFunctionVarArg(fn_type)) { - return; - } - unsigned param_count = LLVMCountParamTypes(fn_type); - LLVMTypeKind return_kind = LLVMGetTypeKind(LLVMGetReturnType(fn_type)); - bool is_memmove = name == "memmove" && param_count == 3 && return_kind == LLVMPointerTypeKind; - bool is_minmax = (string_starts_with(name, str_lit("llvm.minnum.")) || - string_starts_with(name, str_lit("llvm.maxnum."))) && - (return_kind == LLVMFloatTypeKind || - return_kind == LLVMDoubleTypeKind); - if (!is_memmove && !is_minmax) { - return; - } - - gbString wrapper_name = gb_string_make(heap_allocator(), "__$fast_isel$"); - wrapper_name = gb_string_append_length(wrapper_name, name.text, name.len); - defer (gb_string_free(wrapper_name)); - - LLVMValueRef wrapper = LLVMGetNamedFunction(m->mod, wrapper_name); - defer (LLVMSetOperand(call, cast(unsigned)LLVMGetNumOperands(call) - 1, wrapper)); - - if (wrapper != nullptr) { - return; - } - - LLVMCallConv cc = cast(LLVMCallConv)LLVMGetFunctionCallConv(callee); - wrapper = LLVMAddFunction(m->mod, wrapper_name, fn_type); - LLVMSetLinkage(wrapper, LLVMInternalLinkage); - LLVMSetFunctionCallConv(wrapper, cc); - lb_add_attribute_to_proc(m, wrapper, "nounwind"); - - LLVMValueRef params[3] = {}; - LLVMGetParams(wrapper, params); - - LLVMPositionBuilderAtEnd(s->builder, LLVMAppendBasicBlockInContext(m->ctx, wrapper, "")); - LLVMSetCurrentDebugLocation2(s->builder, nullptr); - - LLVMValueRef inner = LLVMBuildCall2(s->builder, fn_type, callee, params, param_count, ""); - LLVMSetInstructionCallConv(inner, cc); - LLVMBuildRet(s->builder, inner); -} - -gb_internal bool lb_is_aggregate_type(LLVMTypeRef type) { - LLVMTypeKind kind = LLVMGetTypeKind(type); - return kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind; -} - -gb_internal bool lb_is_plain_access(LLVMValueRef inst) { - return !LLVMGetVolatile(inst) && LLVMGetOrdering(inst) == LLVMAtomicOrderingNotAtomic; -} - -gb_internal void lb_lower_bool_select(lbFastIselLowering *s, LLVMValueRef select) { - LLVMValueRef c = LLVMGetOperand(select, 0); - LLVMValueRef x = LLVMGetOperand(select, 1); - LLVMValueRef y = LLVMGetOperand(select, 2); - if (LLVMGetTypeKind(LLVMTypeOf(c)) != LLVMIntegerTypeKind) { - return; - } - LLVMPositionBuilderBefore(s->builder, select); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(select)); - LLVMValueRef t = LLVMConstInt(LLVMTypeOf(c), 1, false); - - LLVMValueRef res = nullptr; - if (LLVMIsAConstantInt(x)) { - res = LLVMConstIntGetZExtValue(x) ? LLVMBuildOr(s->builder, c, y, "") : LLVMBuildAnd(s->builder, LLVMBuildXor(s->builder, c, t, ""), y, ""); - } else if (LLVMIsAConstantInt(y)) { - res = LLVMConstIntGetZExtValue(y) ? LLVMBuildOr(s->builder, LLVMBuildXor(s->builder, c, t, ""), x, "") : LLVMBuildAnd(s->builder, c, x, ""); - } else { - LLVMValueRef a = LLVMBuildAnd(s->builder, c, x, ""); - LLVMValueRef b = LLVMBuildAnd(s->builder, LLVMBuildXor(s->builder, c, t, ""), y, ""); - res = LLVMBuildOr(s->builder, a, b, ""); - } - LLVMReplaceAllUsesWith(select, res); - LLVMInstructionEraseFromParent(select); -} - -gb_internal void lb_truncate_bool_arguments(lbFastIselLowering *s, LLVMValueRef call) { - LLVMTypeRef i1 = LLVMInt1TypeInContext(s->m->ctx); - LLVMTypeRef i8 = LLVMInt8TypeInContext(s->m->ctx); - LLVMBasicBlockRef block = LLVMGetInstructionParent(call); - unsigned arg_count = LLVMGetNumArgOperands(call); - for (unsigned a = 0; a < arg_count; a++) { - LLVMValueRef arg = LLVMGetOperand(call, a); - if (LLVMTypeOf(arg) != i1 || LLVMIsAConstantInt(arg)) { - continue; - } - LLVMUseRef first_use = LLVMGetFirstUse(arg); - if (LLVMIsATruncInst(arg) && LLVMGetInstructionParent(arg) == block && - first_use != nullptr && LLVMGetNextUse(first_use) == nullptr) { - continue; - } - LLVMPositionBuilderBefore(s->builder, call); - LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(call)); - LLVMValueRef byte = LLVMBuildZExt(s->builder, arg, i8, ""); - LLVMSetOperand(call, a, LLVMBuildTrunc(s->builder, byte, i1, "")); - } -} - -gb_internal void lb_lower_small_switch(lbFastIselLowering *s, LLVMValueRef sw) { - enum {MAX_CASES = 3}; - LLVMValueRef cond = LLVMGetOperand(sw, 0); - unsigned case_count = (cast(unsigned)LLVMGetNumOperands(sw) - 2) / 2; - if (case_count == 0 || case_count > MAX_CASES || LLVMGetIntTypeWidth(LLVMTypeOf(cond)) > 64) { - return; - } - - LLVMBasicBlockRef block = LLVMGetInstructionParent(sw); - LLVMBasicBlockRef next_block = LLVMGetNextBasicBlock(block); - - LLVMValueRef fn = LLVMGetBasicBlockParent(block); - LLVMMetadataRef loc = LLVMInstructionGetDebugLoc(sw); - - LLVMBasicBlockRef from[MAX_CASES+1] = {}; - LLVMBasicBlockRef to [MAX_CASES+1] = {}; - - LLVMBasicBlockRef curr = block; - - LLVMPositionBuilderBefore(s->builder, sw); - LLVMSetCurrentDebugLocation2(s->builder, loc); - - for (unsigned j = 0; j < case_count; j++) { - LLVMBasicBlockRef dest = LLVMGetSuccessor(sw, j+1); - LLVMBasicBlockRef else_block = LLVMGetSwitchDefaultDest(sw); - if (j+1 < case_count) { - if (next_block != nullptr) { - else_block = LLVMInsertBasicBlockInContext(s->m->ctx, next_block, ""); - } else { - else_block = LLVMAppendBasicBlockInContext(s->m->ctx, fn, ""); - } - } - - LLVMValueRef cmp = LLVMBuildICmp(s->builder, LLVMIntEQ, cond, LLVMGetOperand(sw, 2 + 2*j), ""); - LLVMBuildCondBr(s->builder, cmp, dest, else_block); - - from[j] = curr; - to[j] = dest; - - if (j+1 < case_count) { - curr = else_block; - LLVMPositionBuilderAtEnd(s->builder, curr); - LLVMSetCurrentDebugLocation2(s->builder, loc); - } - } - from[case_count] = curr; - to [case_count] = LLVMGetSwitchDefaultDest(sw); - LLVMInstructionEraseFromParent(sw); - - // the incoming entries for `block` become one for each new edge - for (unsigned e = 0; e <= case_count; e++) { - bool seen = false; - for (unsigned k = 0; k < e; k++) { - seen |= to[k] == to[e]; - } - if (seen) { - continue; - } - - LLVMBasicBlockRef dest = to[e]; - for (LLVMValueRef phi = LLVMGetFirstInstruction(dest); phi != nullptr && LLVMIsAPHINode(phi); /**/) { - LLVMValueRef next = LLVMGetNextInstruction(phi); - LLVMValueRef value_from_block = nullptr; - - unsigned incoming_count = LLVMCountIncoming(phi); - for (unsigned k = 0; k < incoming_count; k++) { - if (LLVMGetIncomingBlock(phi, k) == block) { - value_from_block = LLVMGetIncomingValue(phi, k); - } - } - - if (value_from_block != nullptr) { - LLVMPositionBuilderBefore(s->builder, phi); - LLVMSetCurrentDebugLocation2(s->builder, nullptr); - - LLVMValueRef new_phi = LLVMBuildPhi(s->builder, LLVMTypeOf(phi), ""); - for (unsigned k = 0; k < incoming_count; k++) { - LLVMBasicBlockRef incoming_block = LLVMGetIncomingBlock(phi, k); - if (incoming_block != block) { - LLVMValueRef incoming_value = LLVMGetIncomingValue(phi, k); - LLVMAddIncoming(new_phi, &incoming_value, &incoming_block, 1); - } - } - for (unsigned k = 0; k <= case_count; k++) { - if (to[k] == dest) { - LLVMAddIncoming(new_phi, &value_from_block, &from[k], 1); - } - } - LLVMReplaceAllUsesWith(phi, new_phi); - LLVMInstructionEraseFromParent(phi); - } - - phi = next; - } - } -} - -gb_internal void lb_lower_for_fast_isel(lbModule *m) { - lbFastIselLowering s = {}; - s.m = m; - - s.builder = LLVMCreateBuilderInContext(m->ctx); - defer (LLVMDisposeBuilder(s.builder)); - - array_init(&s.phis, heap_allocator()); - defer (array_free(&s.phis)); - - auto work = array_make(heap_allocator(), 0, 64); - defer (array_free(&work)); - - LLVMValueRef last_fn = LLVMGetLastFunction(m->mod); - for (LLVMValueRef fn = LLVMGetFirstFunction(m->mod); fn != nullptr; fn = fn == last_fn ? nullptr : LLVMGetNextFunction(fn)) { - array_clear(&work); - array_clear(&s.phis); - - for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { - for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; i = LLVMGetNextInstruction(i)) { - if (LLVMIsAStoreInst(i) && lb_is_plain_access(i) && lb_is_aggregate_type(LLVMTypeOf(LLVMGetOperand(i, 0)))) { - array_add(&work, i); - } - } - } - for (LLVMValueRef store : work) { - if (!lb_scalarize_aggregate_store(&s, store)) { - lb_lower_large_constant_store(&s, store); - } - } - - // NOTE(bill): A field read out of an aggregate value is read where that aggregate came from - array_clear(&work); - for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { - for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; i = LLVMGetNextInstruction(i)) { - if (LLVMIsAExtractValueInst(i) && !lb_is_aggregate_type(LLVMTypeOf(i))) { - array_add(&work, i); - } else if (LLVMIsASelectInst(i) && LLVMTypeOf(i) == LLVMInt1TypeInContext(m->ctx)) { - array_add(&work, i); - } else if (LLVMIsACallInst(i)) { - array_add(&work, i); - } else if (LLVMIsASwitchInst(i)) { - array_add(&work, i); - } - } - } - for (LLVMValueRef i : work) { - if (LLVMIsASelectInst(i)) { - lb_lower_bool_select(&s, i); - continue; - } - if (LLVMIsACallInst(i)) { - lb_truncate_bool_arguments(&s, i); - lb_redirect_unselectable_call(&s, i); - continue; - } - if (LLVMIsASwitchInst(i)) { - lb_lower_small_switch(&s, i); - continue; - } - LLVMValueRef agg = LLVMGetOperand(i, 0); - unsigned n = LLVMGetNumIndices(i); - if (n > LB_SCALARIZE_MAX_DEPTH) { - continue; - } - - if (!(LLVMIsAConstant(agg) || - LLVMIsAInsertValueInst(agg) || - LLVMIsASelectInst(agg) || - LLVMIsAPHINode(agg) || - (LLVMIsALoadInst(agg) && lb_is_plain_access(agg)))) { - continue; - } - - s.store = i; - - LLVMValueRef field = nullptr; - if (lb_aggregate_field_value(&s, agg, LLVMGetIndices(i), n, &field) && field != nullptr) { - LLVMReplaceAllUsesWith(i, field); - LLVMInstructionEraseFromParent(i); - } - } - - bool removed = true; - while (removed) { - removed = false; - for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { - for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; /**/) { - LLVMValueRef next = LLVMGetNextInstruction(i); - if (LLVMGetFirstUse(i) == nullptr && - (LLVMIsAInsertValueInst(i) || LLVMIsAExtractValueInst(i) || LLVMIsASelectInst(i) || LLVMIsAPHINode(i) || - LLVMIsAGetElementPtrInst(i) || (LLVMIsALoadInst(i) && lb_is_plain_access(i)))) { - LLVMInstructionEraseFromParent(i); - removed = true; - } - i = next; - } - } - } - } -} - struct lbLLVMEmitWorker { LLVMTargetMachineRef target_machine; LLVMCodeGenFileType code_gen_file_type; diff --git a/src/llvm_backend.hpp b/src/llvm_backend.hpp index 7a3b12575..5bc9c95ff 100644 --- a/src/llvm_backend.hpp +++ b/src/llvm_backend.hpp @@ -630,6 +630,9 @@ gb_internal void lb_mem_copy_non_overlapping(lbProcedure *p, lbValue dst, lbValu gb_internal LLVMValueRef lb_mem_zero_ptr_internal(lbProcedure *p, LLVMValueRef ptr, LLVMValueRef len, unsigned alignment, bool is_volatile); gb_internal LLVMValueRef lb_mem_zero_ptr_internal(lbProcedure *p, LLVMValueRef ptr, usize len, unsigned alignment, bool is_volatile); +gb_internal bool lb_const_has_misaligned_pointer(LLVMTargetDataRef td, LLVMValueRef c, u64 offset, u64 base_align); +gb_internal void lb_add_attribute_to_proc(lbModule *m, LLVMValueRef proc_value, char const *name, u64 value=0); + gb_internal gb_inline i64 lb_max_zero_init_size(void) { if (build_context.metrics.os == TargetOs_darwin && build_context.metrics.arch == TargetArch_arm64) { // https://github.com/odin-lang/Odin/issues/6347 diff --git a/src/llvm_backend_general.cpp b/src/llvm_backend_general.cpp index 9893debbf..5d26220ac 100644 --- a/src/llvm_backend_general.cpp +++ b/src/llvm_backend_general.cpp @@ -3514,7 +3514,7 @@ gb_internal void lb_add_nocapture_proc_attribute_at_index(lbProcedure *p, isize LLVMAddAttributeAtIndex(p->value, cast(unsigned)index, lb_create_nocapture_attribute(p->module->ctx)); } -gb_internal void lb_add_attribute_to_proc(lbModule *m, LLVMValueRef proc_value, char const *name, u64 value=0) { +gb_internal void lb_add_attribute_to_proc(lbModule *m, LLVMValueRef proc_value, char const *name, u64 value) { LLVMAddAttributeAtIndex(proc_value, LLVMAttributeIndex_FunctionIndex, lb_create_enum_attribute(m->ctx, name, value)); } diff --git a/src/llvm_backend_opt.cpp b/src/llvm_backend_opt.cpp index 0f925301c..4df0cc27b 100644 --- a/src/llvm_backend_opt.cpp +++ b/src/llvm_backend_opt.cpp @@ -488,4 +488,623 @@ gb_internal void lb_run_remove_unused_globals_pass(lbModule *m) { } } +// NOTE(bill, 2026-10-03) +// +// LLVM's fast instruction selector cannot select a first class aggregate `load`, `store`, `insertvalue`, or `select`, +// and hands the rest of the block to SelectionDAG. +// SROA leaves many behind so it stores/loads the fields instead +enum { + LB_SCALARIZE_MAX_LEAVES = 32, + LB_SCALARIZE_MAX_DEPTH = 8, +}; +struct lbAggregateLeaf { + unsigned path[LB_SCALARIZE_MAX_DEPTH]; + unsigned depth; +}; + +struct lbScalarizedPhi { + LLVMValueRef aggregate; + unsigned path[LB_SCALARIZE_MAX_DEPTH]; + unsigned depth; + LLVMValueRef field; + bool ok; +}; + +struct lbFastIselLowering { + lbModule * m; + LLVMBuilderRef builder; + LLVMValueRef store; + Array phis; +}; + +gb_internal i64 lb_aggregate_path_offset(LLVMTypeRef type, unsigned const *path, unsigned depth, LLVMTypeRef *leaf_type_) { + i64 offset = 0; + for (unsigned d = 0; d < depth; d++) { + if (LLVMGetTypeKind(type) == LLVMStructTypeKind) { + bool is_packed = LLVMIsPackedStruct(type); + i64 field_offset = 0; + for (unsigned i = 0; i <= path[d]; i++) { + LLVMTypeRef field = LLVMStructGetTypeAtIndex(type, i); + if (!is_packed) { + field_offset = llvm_align_formula(field_offset, lb_alignof(field)); + } + if (i == path[d]) { + type = field; + break; + } + field_offset += lb_sizeof(field); + } + offset += field_offset; + } else { + type = OdinLLVMGetArrayElementType(type); + offset += cast(i64)path[d] * lb_sizeof(type); + } + } + if (leaf_type_) *leaf_type_ = type; + return offset; +} + +gb_internal bool lb_aggregate_leaves(LLVMTypeRef type, unsigned *path, unsigned depth, lbAggregateLeaf *leaves, isize *leaf_count) { + LLVMTypeKind kind = LLVMGetTypeKind(type); + if (kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind) { + if (depth >= LB_SCALARIZE_MAX_DEPTH) { + return false; + } + if (kind == LLVMStructTypeKind && LLVMIsOpaqueStruct(type)) { + return false; + } + unsigned count = kind == LLVMStructTypeKind ? LLVMCountStructElementTypes(type) : cast(unsigned)LLVMGetArrayLength(type); + for (unsigned i = 0; i < count; i++) { + path[depth] = i; + LLVMTypeRef elem = kind == LLVMStructTypeKind ? LLVMStructGetTypeAtIndex(type, i) : OdinLLVMGetArrayElementType(type); + if (!lb_aggregate_leaves(elem, path, depth+1, leaves, leaf_count)) { + return false; + } + } + return true; + } + + if (*leaf_count >= LB_SCALARIZE_MAX_LEAVES) { + return false; + } + + lbAggregateLeaf *leaf = &leaves[(*leaf_count)++]; + gb_memmove(leaf->path, path, depth*gb_size_of(unsigned)); + leaf->depth = depth; + return true; +} + +gb_internal unsigned lb_aggregate_field_alignment(unsigned alignment, i64 offset) { + unsigned a = gb_max(alignment, 1u); + while (offset % a != 0) { + a >>= 1; + } + return a; +} + +gb_internal LLVMValueRef lb_aggregate_field_gep(lbModule *m, LLVMBuilderRef b, LLVMTypeRef type, LLVMValueRef ptr, unsigned const *path, unsigned depth) { + LLVMValueRef indices[LB_SCALARIZE_MAX_DEPTH+1] = {}; + LLVMTypeRef i32 = LLVMInt32TypeInContext(m->ctx); + indices[0] = LLVMConstInt(i32, 0, false); + for (unsigned d = 0; d < depth; d++) { + indices[d+1] = LLVMConstInt(i32, path[d], false); + } + return LLVMBuildInBoundsGEP2(b, type, ptr, indices, depth+1, ""); +} + +// returns false if the field cannot be reached without an aggregate value, and sets `*field_` to nullptr when it is undefined +gb_internal bool lb_aggregate_field_value(lbFastIselLowering *s, LLVMValueRef v, unsigned const *path, unsigned depth, LLVMValueRef *field_) { + if (depth == 0) { + *field_ = (LLVMIsUndef(v) || LLVMIsPoison(v)) ? nullptr : v; + return true; + } + if (LLVMIsUndef(v) || LLVMIsPoison(v)) { + *field_ = nullptr; + return true; + } + + if (LLVMIsAConstant(v)) { + LLVMValueRef elem = LLVMGetAggregateElement(v, path[0]); + if (elem != nullptr) { + return lb_aggregate_field_value(s, elem, path+1, depth-1, field_); + } + } else if (LLVMIsAInsertValueInst(v)) { + unsigned n = LLVMGetNumIndices(v); + unsigned const *indices = LLVMGetIndices(v); + + unsigned k = 0; + while (k < n && k < depth && indices[k] == path[k]) { + k += 1; + } + + if (k == n) { + return lb_aggregate_field_value(s, LLVMGetOperand(v, 1), path+n, depth-n, field_); + } else if (k < depth) { + return lb_aggregate_field_value(s, LLVMGetOperand(v, 0), path, depth, field_); + } + return false; + } else if (LLVMIsAExtractValueInst(v)) { + unsigned n = LLVMGetNumIndices(v); + if (n + depth <= LB_SCALARIZE_MAX_DEPTH) { + unsigned full[LB_SCALARIZE_MAX_DEPTH]; + gb_memmove(full, LLVMGetIndices(v), n*gb_size_of(unsigned)); + gb_memmove(full+n, path, depth*gb_size_of(unsigned)); + return lb_aggregate_field_value(s, LLVMGetOperand(v, 0), full, n+depth, field_); + } + return false; + } else if (LLVMIsASelectInst(v)) { + LLVMValueRef x = nullptr; + LLVMValueRef y = nullptr; + if (!lb_aggregate_field_value(s, LLVMGetOperand(v, 1), path, depth, &x) || + !lb_aggregate_field_value(s, LLVMGetOperand(v, 2), path, depth, &y)) { + return false; + } + if (x == nullptr || y == nullptr) { + *field_ = x ? x : y; + return true; + } + LLVMPositionBuilderBefore(s->builder, s->store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(s->store)); + *field_ = LLVMBuildSelect(s->builder, LLVMGetOperand(v, 0), x, y, ""); + return true; + } else if (LLVMIsAPHINode(v)) { + for (lbScalarizedPhi const &e : s->phis) { + if (e.aggregate == v && + e.depth == depth && + gb_memcompare(e.path, path, depth*gb_size_of(unsigned)) == 0) { + *field_ = e.field; + return e.ok; + } + } + + LLVMTypeRef field_type = nullptr; + lb_aggregate_path_offset(LLVMTypeOf(v), path, depth, &field_type); + + LLVMPositionBuilderBefore(s->builder, v); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + + // NOTE(bill): cached before its incoming values, as a loop reaches it again + isize index = s->phis.count; + lbScalarizedPhi entry = {}; + entry.aggregate = v; + gb_memmove(entry.path, path, depth*gb_size_of(unsigned)); + + LLVMValueRef phi = LLVMBuildPhi(s->builder, field_type, ""); + entry.depth = depth; + entry.field = phi; + entry.ok = true; + array_add(&s->phis, entry); + + LLVMValueRef saved_store = s->store; + bool ok = true; + unsigned incoming_count = LLVMCountIncoming(v); + for (unsigned k = 0; k < incoming_count; k++) { + LLVMBasicBlockRef block = LLVMGetIncomingBlock(v, k); + s->store = LLVMGetBasicBlockTerminator(block); + LLVMValueRef field = nullptr; + if (!lb_aggregate_field_value(s, LLVMGetIncomingValue(v, k), path, depth, &field)) { + ok = false; + field = nullptr; + } + if (field == nullptr) { + field = LLVMGetUndef(field_type); + } + LLVMAddIncoming(phi, &field, &block, 1); + } + s->store = saved_store; + s->phis[index].ok = ok; + *field_ = phi; + return ok; + } else if (LLVMIsALoadInst(v) && !LLVMGetVolatile(v) && LLVMGetOrdering(v) == LLVMAtomicOrderingNotAtomic) { + // NOTE(bill): Read the field where the aggregate was read, as the memory may change before the store + LLVMTypeRef type = LLVMTypeOf(v); + LLVMTypeRef field_type = nullptr; + i64 offset = lb_aggregate_path_offset(type, path, depth, &field_type); + + LLVMPositionBuilderBefore(s->builder, LLVMGetNextInstruction(v)); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(v)); + + LLVMValueRef ptr = lb_aggregate_field_gep(s->m, s->builder, type, LLVMGetOperand(v, 0), path, depth); + LLVMValueRef field = LLVMBuildLoad2(s->builder, field_type, ptr, ""); + LLVMSetAlignment(field, lb_aggregate_field_alignment(LLVMGetAlignment(v), offset)); + + *field_ = field; + return true; + } + + if (depth == 1) { + LLVMPositionBuilderBefore(s->builder, s->store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(s->store)); + *field_ = LLVMBuildExtractValue(s->builder, v, path[0], ""); + return true; + } + return false; +} + +gb_internal bool lb_scalarize_aggregate_store(lbFastIselLowering *s, LLVMValueRef store) { + LLVMValueRef value = LLVMGetOperand(store, 0); + LLVMValueRef ptr = LLVMGetOperand(store, 1); + LLVMTypeRef type = LLVMTypeOf(value); + + lbAggregateLeaf leaves[LB_SCALARIZE_MAX_LEAVES] = {}; + isize leaf_count = 0; + unsigned path[LB_SCALARIZE_MAX_DEPTH] = {}; + + if (!lb_aggregate_leaves(type, path, 0, leaves, &leaf_count)) { + return false; + } + s->store = store; + + LLVMValueRef fields[LB_SCALARIZE_MAX_LEAVES] = {}; + for (isize i = 0; i < leaf_count; i++) { + if (!lb_aggregate_field_value(s, value, leaves[i].path, leaves[i].depth, &fields[i])) { + return false; + } + } + + unsigned alignment = LLVMGetAlignment(store); + + LLVMPositionBuilderBefore(s->builder, store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); + + for (isize i = 0; i < leaf_count; i++) { + if (fields[i] == nullptr) { + continue; + } + i64 offset = lb_aggregate_path_offset(type, leaves[i].path, leaves[i].depth, nullptr); + LLVMValueRef field_ptr = lb_aggregate_field_gep(s->m, s->builder, type, ptr, leaves[i].path, leaves[i].depth); + LLVMValueRef field_store = LLVMBuildStore(s->builder, fields[i], field_ptr); + LLVMSetAlignment(field_store, lb_aggregate_field_alignment(alignment, offset)); + } + LLVMInstructionEraseFromParent(store); + return true; +} + +// NOTE(bill): If a a constant has too many fields to store one at a time is set or copied from a constant instead +gb_internal void lb_lower_large_constant_store(lbFastIselLowering *s, LLVMValueRef store) { + lbModule *m = s->m; + LLVMValueRef value = LLVMGetOperand(store, 0); + LLVMValueRef ptr = LLVMGetOperand(store, 1); + LLVMTypeRef type = LLVMTypeOf(value); + + LLVMTargetDataRef data_layout = LLVMGetModuleDataLayout(m->mod); + unsigned alignment = gb_max(LLVMGetAlignment(store), 1u); + if (!LLVMIsAConstant(value) || lb_const_has_misaligned_pointer(data_layout, value, 0, alignment)) { + return; + } + + LLVMPositionBuilderBefore(s->builder, store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); + + LLVMValueRef size = LLVMConstInt(LLVMInt64TypeInContext(m->ctx), LLVMStoreSizeOfType(data_layout, type), false); + if (LLVMIsNull(value)) { + LLVMBuildMemSet(s->builder, ptr, LLVMConstInt(LLVMInt8TypeInContext(m->ctx), 0, false), size, alignment); + LLVMInstructionEraseFromParent(store); + return; + } + + LLVMValueRef global = LLVMAddGlobal(m->mod, type, ""); + LLVMSetInitializer(global, value); + LLVMSetGlobalConstant(global, true); + LLVMSetAlignment(global, alignment); + + LLVMSetLinkage(global, LLVMPrivateLinkage); + LLVMSetUnnamedAddress(global, LLVMGlobalUnnamedAddr); + + LLVMBuildMemCpy(s->builder, ptr, alignment, global, alignment, size); + + LLVMInstructionEraseFromParent(store); +} + +// NOTE(bill): The fast instruction selector in LLVM cannot select a call to `memmove` nor to the floating point `minnum` and `maxnum`. +// To improve things we can them through a function of the module which calls them +gb_internal void lb_redirect_unselectable_call(lbFastIselLowering *s, LLVMValueRef call) { + lbModule *m = s->m; + + LLVMValueRef callee = LLVMGetCalledValue(call); + if (!LLVMIsAFunction(callee)) { + return; + } + size_t name_len = 0; + char const *name_text = LLVMGetValueName2(callee, &name_len); + String name = make_string(cast(u8 const *)name_text, name_len); + + LLVMTypeRef fn_type = LLVMGlobalGetValueType(callee); + if (LLVMGetCalledFunctionType(call) != fn_type || LLVMIsFunctionVarArg(fn_type)) { + return; + } + unsigned param_count = LLVMCountParamTypes(fn_type); + LLVMTypeKind return_kind = LLVMGetTypeKind(LLVMGetReturnType(fn_type)); + bool is_memmove = name == "memmove" && param_count == 3 && return_kind == LLVMPointerTypeKind; + bool is_minmax = (string_starts_with(name, str_lit("llvm.minnum.")) || + string_starts_with(name, str_lit("llvm.maxnum."))) && + (return_kind == LLVMFloatTypeKind || + return_kind == LLVMDoubleTypeKind); + if (!is_memmove && !is_minmax) { + return; + } + + gbString wrapper_name = gb_string_make(heap_allocator(), "__$fast_isel$"); + wrapper_name = gb_string_append_length(wrapper_name, name.text, name.len); + defer (gb_string_free(wrapper_name)); + + LLVMValueRef wrapper = LLVMGetNamedFunction(m->mod, wrapper_name); + defer (LLVMSetOperand(call, cast(unsigned)LLVMGetNumOperands(call) - 1, wrapper)); + + if (wrapper != nullptr) { + return; + } + + LLVMCallConv cc = cast(LLVMCallConv)LLVMGetFunctionCallConv(callee); + wrapper = LLVMAddFunction(m->mod, wrapper_name, fn_type); + LLVMSetLinkage(wrapper, LLVMInternalLinkage); + LLVMSetFunctionCallConv(wrapper, cc); + lb_add_attribute_to_proc(m, wrapper, "nounwind"); + + LLVMValueRef params[3] = {}; + LLVMGetParams(wrapper, params); + + LLVMPositionBuilderAtEnd(s->builder, LLVMAppendBasicBlockInContext(m->ctx, wrapper, "")); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + LLVMValueRef inner = LLVMBuildCall2(s->builder, fn_type, callee, params, param_count, ""); + LLVMSetInstructionCallConv(inner, cc); + LLVMBuildRet(s->builder, inner); +} + +gb_internal bool lb_is_aggregate_type(LLVMTypeRef type) { + LLVMTypeKind kind = LLVMGetTypeKind(type); + return kind == LLVMStructTypeKind || kind == LLVMArrayTypeKind; +} + +gb_internal bool lb_is_plain_access(LLVMValueRef inst) { + return !LLVMGetVolatile(inst) && LLVMGetOrdering(inst) == LLVMAtomicOrderingNotAtomic; +} + +gb_internal void lb_lower_bool_select(lbFastIselLowering *s, LLVMValueRef select) { + LLVMValueRef c = LLVMGetOperand(select, 0); + LLVMValueRef x = LLVMGetOperand(select, 1); + LLVMValueRef y = LLVMGetOperand(select, 2); + if (LLVMGetTypeKind(LLVMTypeOf(c)) != LLVMIntegerTypeKind) { + return; + } + LLVMPositionBuilderBefore(s->builder, select); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(select)); + LLVMValueRef t = LLVMConstInt(LLVMTypeOf(c), 1, false); + + LLVMValueRef res = nullptr; + if (LLVMIsAConstantInt(x)) { + res = LLVMConstIntGetZExtValue(x) ? LLVMBuildOr(s->builder, c, y, "") : LLVMBuildAnd(s->builder, LLVMBuildXor(s->builder, c, t, ""), y, ""); + } else if (LLVMIsAConstantInt(y)) { + res = LLVMConstIntGetZExtValue(y) ? LLVMBuildOr(s->builder, LLVMBuildXor(s->builder, c, t, ""), x, "") : LLVMBuildAnd(s->builder, c, x, ""); + } else { + LLVMValueRef a = LLVMBuildAnd(s->builder, c, x, ""); + LLVMValueRef b = LLVMBuildAnd(s->builder, LLVMBuildXor(s->builder, c, t, ""), y, ""); + res = LLVMBuildOr(s->builder, a, b, ""); + } + LLVMReplaceAllUsesWith(select, res); + LLVMInstructionEraseFromParent(select); +} + +gb_internal void lb_truncate_bool_arguments(lbFastIselLowering *s, LLVMValueRef call) { + LLVMTypeRef i1 = LLVMInt1TypeInContext(s->m->ctx); + LLVMTypeRef i8 = LLVMInt8TypeInContext(s->m->ctx); + LLVMBasicBlockRef block = LLVMGetInstructionParent(call); + unsigned arg_count = LLVMGetNumArgOperands(call); + for (unsigned a = 0; a < arg_count; a++) { + LLVMValueRef arg = LLVMGetOperand(call, a); + if (LLVMTypeOf(arg) != i1 || LLVMIsAConstantInt(arg)) { + continue; + } + LLVMUseRef first_use = LLVMGetFirstUse(arg); + if (LLVMIsATruncInst(arg) && LLVMGetInstructionParent(arg) == block && + first_use != nullptr && LLVMGetNextUse(first_use) == nullptr) { + continue; + } + LLVMPositionBuilderBefore(s->builder, call); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(call)); + LLVMValueRef byte = LLVMBuildZExt(s->builder, arg, i8, ""); + LLVMSetOperand(call, a, LLVMBuildTrunc(s->builder, byte, i1, "")); + } +} + +gb_internal void lb_lower_small_switch(lbFastIselLowering *s, LLVMValueRef sw) { + enum {MAX_CASES = 3}; + LLVMValueRef cond = LLVMGetOperand(sw, 0); + unsigned case_count = (cast(unsigned)LLVMGetNumOperands(sw) - 2) / 2; + if (case_count == 0 || case_count > MAX_CASES || LLVMGetIntTypeWidth(LLVMTypeOf(cond)) > 64) { + return; + } + + LLVMBasicBlockRef block = LLVMGetInstructionParent(sw); + LLVMBasicBlockRef next_block = LLVMGetNextBasicBlock(block); + + LLVMValueRef fn = LLVMGetBasicBlockParent(block); + LLVMMetadataRef loc = LLVMInstructionGetDebugLoc(sw); + + LLVMBasicBlockRef from[MAX_CASES+1] = {}; + LLVMBasicBlockRef to [MAX_CASES+1] = {}; + + LLVMBasicBlockRef curr = block; + + LLVMPositionBuilderBefore(s->builder, sw); + LLVMSetCurrentDebugLocation2(s->builder, loc); + + for (unsigned j = 0; j < case_count; j++) { + LLVMBasicBlockRef dest = LLVMGetSuccessor(sw, j+1); + LLVMBasicBlockRef else_block = LLVMGetSwitchDefaultDest(sw); + if (j+1 < case_count) { + if (next_block != nullptr) { + else_block = LLVMInsertBasicBlockInContext(s->m->ctx, next_block, ""); + } else { + else_block = LLVMAppendBasicBlockInContext(s->m->ctx, fn, ""); + } + } + + LLVMValueRef cmp = LLVMBuildICmp(s->builder, LLVMIntEQ, cond, LLVMGetOperand(sw, 2 + 2*j), ""); + LLVMBuildCondBr(s->builder, cmp, dest, else_block); + + from[j] = curr; + to[j] = dest; + + if (j+1 < case_count) { + curr = else_block; + LLVMPositionBuilderAtEnd(s->builder, curr); + LLVMSetCurrentDebugLocation2(s->builder, loc); + } + } + from[case_count] = curr; + to [case_count] = LLVMGetSwitchDefaultDest(sw); + LLVMInstructionEraseFromParent(sw); + + // the incoming entries for `block` become one for each new edge + for (unsigned e = 0; e <= case_count; e++) { + bool seen = false; + for (unsigned k = 0; k < e; k++) { + seen |= to[k] == to[e]; + } + if (seen) { + continue; + } + + LLVMBasicBlockRef dest = to[e]; + for (LLVMValueRef phi = LLVMGetFirstInstruction(dest); phi != nullptr && LLVMIsAPHINode(phi); /**/) { + LLVMValueRef next = LLVMGetNextInstruction(phi); + LLVMValueRef value_from_block = nullptr; + + unsigned incoming_count = LLVMCountIncoming(phi); + for (unsigned k = 0; k < incoming_count; k++) { + if (LLVMGetIncomingBlock(phi, k) == block) { + value_from_block = LLVMGetIncomingValue(phi, k); + } + } + + if (value_from_block != nullptr) { + LLVMPositionBuilderBefore(s->builder, phi); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + LLVMValueRef new_phi = LLVMBuildPhi(s->builder, LLVMTypeOf(phi), ""); + for (unsigned k = 0; k < incoming_count; k++) { + LLVMBasicBlockRef incoming_block = LLVMGetIncomingBlock(phi, k); + if (incoming_block != block) { + LLVMValueRef incoming_value = LLVMGetIncomingValue(phi, k); + LLVMAddIncoming(new_phi, &incoming_value, &incoming_block, 1); + } + } + for (unsigned k = 0; k <= case_count; k++) { + if (to[k] == dest) { + LLVMAddIncoming(new_phi, &value_from_block, &from[k], 1); + } + } + LLVMReplaceAllUsesWith(phi, new_phi); + LLVMInstructionEraseFromParent(phi); + } + + phi = next; + } + } +} + +gb_internal void lb_lower_for_fast_isel(lbModule *m) { + lbFastIselLowering s = {}; + s.m = m; + + s.builder = LLVMCreateBuilderInContext(m->ctx); + defer (LLVMDisposeBuilder(s.builder)); + + array_init(&s.phis, heap_allocator()); + defer (array_free(&s.phis)); + + auto work = array_make(heap_allocator(), 0, 64); + defer (array_free(&work)); + + LLVMValueRef last_fn = LLVMGetLastFunction(m->mod); + for (LLVMValueRef fn = LLVMGetFirstFunction(m->mod); fn != nullptr; fn = fn == last_fn ? nullptr : LLVMGetNextFunction(fn)) { + array_clear(&work); + array_clear(&s.phis); + + for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { + for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; i = LLVMGetNextInstruction(i)) { + if (LLVMIsAStoreInst(i) && lb_is_plain_access(i) && lb_is_aggregate_type(LLVMTypeOf(LLVMGetOperand(i, 0)))) { + array_add(&work, i); + } + } + } + for (LLVMValueRef store : work) { + if (!lb_scalarize_aggregate_store(&s, store)) { + lb_lower_large_constant_store(&s, store); + } + } + + // NOTE(bill): A field read out of an aggregate value is read where that aggregate came from + array_clear(&work); + for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { + for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; i = LLVMGetNextInstruction(i)) { + if (LLVMIsAExtractValueInst(i) && !lb_is_aggregate_type(LLVMTypeOf(i))) { + array_add(&work, i); + } else if (LLVMIsASelectInst(i) && LLVMTypeOf(i) == LLVMInt1TypeInContext(m->ctx)) { + array_add(&work, i); + } else if (LLVMIsACallInst(i)) { + array_add(&work, i); + } else if (LLVMIsASwitchInst(i)) { + array_add(&work, i); + } + } + } + for (LLVMValueRef i : work) { + if (LLVMIsASelectInst(i)) { + lb_lower_bool_select(&s, i); + continue; + } + if (LLVMIsACallInst(i)) { + lb_truncate_bool_arguments(&s, i); + lb_redirect_unselectable_call(&s, i); + continue; + } + if (LLVMIsASwitchInst(i)) { + lb_lower_small_switch(&s, i); + continue; + } + LLVMValueRef agg = LLVMGetOperand(i, 0); + unsigned n = LLVMGetNumIndices(i); + if (n > LB_SCALARIZE_MAX_DEPTH) { + continue; + } + + if (!(LLVMIsAConstant(agg) || + LLVMIsAInsertValueInst(agg) || + LLVMIsASelectInst(agg) || + LLVMIsAPHINode(agg) || + (LLVMIsALoadInst(agg) && lb_is_plain_access(agg)))) { + continue; + } + + s.store = i; + + LLVMValueRef field = nullptr; + if (lb_aggregate_field_value(&s, agg, LLVMGetIndices(i), n, &field) && field != nullptr) { + LLVMReplaceAllUsesWith(i, field); + LLVMInstructionEraseFromParent(i); + } + } + + bool removed = true; + while (removed) { + removed = false; + for (LLVMBasicBlockRef bb = LLVMGetFirstBasicBlock(fn); bb != nullptr; bb = LLVMGetNextBasicBlock(bb)) { + for (LLVMValueRef i = LLVMGetFirstInstruction(bb); i != nullptr; /**/) { + LLVMValueRef next = LLVMGetNextInstruction(i); + if (LLVMGetFirstUse(i) == nullptr && + (LLVMIsAInsertValueInst(i) || LLVMIsAExtractValueInst(i) || LLVMIsASelectInst(i) || LLVMIsAPHINode(i) || + LLVMIsAGetElementPtrInst(i) || (LLVMIsALoadInst(i) && lb_is_plain_access(i)))) { + LLVMInstructionEraseFromParent(i); + removed = true; + } + i = next; + } + } + } + } +} From c7de0b9e36195f5b6ae78a84edb7b087070f71a3 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 10:48:53 +0100 Subject: [PATCH 17/20] Copy a constant built in a local of its own rather than loading and storing it, and use a constant slice's backing array directly --- src/llvm_backend_const.cpp | 41 ++++++++++++++++++++++---------------- 1 file changed, 24 insertions(+), 17 deletions(-) diff --git a/src/llvm_backend_const.cpp b/src/llvm_backend_const.cpp index 7eda9515d..b7b2a65f2 100644 --- a/src/llvm_backend_const.cpp +++ b/src/llvm_backend_const.cpp @@ -524,6 +524,21 @@ gb_internal lbValue lb_emit_source_code_location_as_global(lbProcedure *p, Ast * +// NOTE(bill): Constants which cannot be an LLVM constant were built in a local of their own +// which was copied from rather than loaded and stored as a first class aggregate +gb_internal void lb_store_local_constant(lbProcedure *p, LLVMValueRef dst, LLVMValueRef value) { + if (!LLVMIsALoadInst(value) || !LLVMIsAAllocaInst(LLVMGetOperand(value, 0))) { + LLVMBuildStore(p->builder, value, dst); + return; + } + LLVMValueRef src = LLVMGetOperand(value, 0); + LLVMTypeRef type = LLVMTypeOf(value); + LLVMValueRef size = LLVMConstInt(lb_type(p->module, t_int), lb_sizeof(type), false); + + unsigned dst_alignment = LLVMABIAlignmentOfType(LLVMGetModuleDataLayout(p->module->mod), type); + LLVMBuildMemCpy(p->builder, dst, dst_alignment, src, lb_try_get_alignment(src, 1), size); +} + gb_internal LLVMValueRef lb_build_constant_array_values(lbModule *m, Type *type, Type *elem_type, isize count, LLVMValueRef *values, lbConstContext cc) { if (cc.allow_local) { cc.is_rodata = false; @@ -552,7 +567,7 @@ gb_internal LLVMValueRef lb_build_constant_array_values(lbModule *m, Type *type, if (is_type_proc(elem_type)) { values[i] = LLVMConstPointerCast(values[i], llvm_elem_type); } - LLVMBuildStore(p->builder, values[i], elem.value); + lb_store_local_constant(p, elem.value, values[i]); } return lb_addr_load(p, v).value; } @@ -1141,6 +1156,12 @@ gb_internal lbValue lb_const_value(lbModule *m, Type *type, ExactValue value, lb local_copy, alignment, LLVMConstInt(lb_type(m, t_int), type_size_of(t), false) ); + } else if (LLVMIsALoadInst(backing_array.value) && LLVMIsAAllocaInst(LLVMGetOperand(backing_array.value, 0)) && + LLVMGetFirstUse(backing_array.value) == nullptr) { + // NOTE(bill): the backing data was built in a local of its own which the slice uses rather than a copy of it + array_data = LLVMGetOperand(backing_array.value, 0); + LLVMSetAlignment(array_data, gb_max(LLVMGetAlignment(array_data), alignment)); + LLVMInstructionEraseFromParent(backing_array.value); } else { array_data = llvm_alloca(p, LLVMTypeOf(backing_array.value), alignment); LLVMBuildStore(p->builder, backing_array.value, array_data); @@ -2166,13 +2187,7 @@ gb_internal lbValue lb_const_value(lbModule *m, Type *type, ExactValue value, lb LLVMValueRef dst = LLVMBuildGEP2(p->builder, field_llvm_type, ptr, indices, idx_list_len+1, ""); dst = LLVMBuildPointerCast(p->builder, dst, lb_type(m, alloc_type_pointer(tav.type)), ""); - if (LLVMIsALoadInst(elem_value)) { - i64 sz = type_size_of(tav.type); - LLVMValueRef src = LLVMGetOperand(elem_value, 0); - lb_mem_copy_non_overlapping(p, {dst, t_rawptr}, {src, t_rawptr}, lb_const_int(m, t_int, sz), false); - } else { - LLVMBuildStore(p->builder, elem_value, dst); - } + lb_store_local_constant(p, dst, elem_value); values[index] = LLVMBuildLoad2(p->builder, field_llvm_type, ptr, ""); @@ -2262,15 +2277,7 @@ gb_internal lbValue lb_const_value(lbModule *m, Type *type, ExactValue value, lb LLVMValueRef val = old_values[i]; if (!LLVMIsConstant(val)) { LLVMValueRef dst = LLVMBuildStructGEP2(p->builder, llvm_addr_type(p->module, v.addr), v.addr.value, cast(unsigned)i, ""); - // if (LLVMIsALoadInst(val)) { - // Type *ptr_type = v.addr.type; - // i64 sz = type_size_of(type_deref(ptr_type)); - - // LLVMValueRef src = LLVMGetOperand(val, 0); - // lb_mem_copy_non_overlapping(p, {dst, ptr_type}, {src, ptr_type}, lb_const_int(m, t_int, sz), false); - // } else { - LLVMBuildStore(p->builder, val, dst); - // } + lb_store_local_constant(p, dst, val); } } return lb_addr_load(p, v); From 52fcaaf024a8fd75cf2abe784679610d187a5157 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 10:49:41 +0100 Subject: [PATCH 18/20] Copy a large loaded aggregate through a local when the fast instruction selector cannot store it one field at a time --- src/llvm_backend_opt.cpp | 43 +++++++++++++++++++++++++++++++++++++++- 1 file changed, 42 insertions(+), 1 deletion(-) diff --git a/src/llvm_backend_opt.cpp b/src/llvm_backend_opt.cpp index 4df0cc27b..d3502a147 100644 --- a/src/llvm_backend_opt.cpp +++ b/src/llvm_backend_opt.cpp @@ -863,6 +863,42 @@ gb_internal bool lb_is_plain_access(LLVMValueRef inst) { return !LLVMGetVolatile(inst) && LLVMGetOrdering(inst) == LLVMAtomicOrderingNotAtomic; } +gb_internal void lb_lower_large_loaded_store(lbFastIselLowering *s, LLVMValueRef store) { + lbModule *m = s->m; + LLVMValueRef load = LLVMGetOperand(store, 0); + LLVMValueRef ptr = LLVMGetOperand(store, 1); + LLVMUseRef use = LLVMGetFirstUse(load); + if (!lb_is_plain_access(load) || LLVMGetNextUse(use) != nullptr) { + return; + } + + LLVMTypeRef type = LLVMTypeOf(load); + LLVMTargetDataRef data_layout = LLVMGetModuleDataLayout(m->mod); + LLVMValueRef size = LLVMConstInt(LLVMInt64TypeInContext(m->ctx), LLVMStoreSizeOfType(data_layout, type), false); + + unsigned load_alignment = gb_max(LLVMGetAlignment(load), 1u); + unsigned store_alignment = gb_max(LLVMGetAlignment(store), 1u); + unsigned temp_alignment = gb_max(LLVMABIAlignmentOfType(data_layout, type), load_alignment); + + LLVMValueRef fn = LLVMGetBasicBlockParent(LLVMGetInstructionParent(store)); + LLVMPositionBuilderBefore(s->builder, LLVMGetFirstInstruction(LLVMGetEntryBasicBlock(fn))); + LLVMSetCurrentDebugLocation2(s->builder, nullptr); + + LLVMValueRef temp = LLVMBuildAlloca(s->builder, type, ""); + LLVMSetAlignment(temp, temp_alignment); + + LLVMPositionBuilderBefore(s->builder, LLVMGetNextInstruction(load)); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(load)); + LLVMBuildMemCpy(s->builder, temp, temp_alignment, LLVMGetOperand(load, 0), load_alignment, size); + + LLVMPositionBuilderBefore(s->builder, store); + LLVMSetCurrentDebugLocation2(s->builder, LLVMInstructionGetDebugLoc(store)); + LLVMBuildMemCpy(s->builder, ptr, store_alignment, temp, temp_alignment, size); + + LLVMInstructionEraseFromParent(store); + LLVMInstructionEraseFromParent(load); +} + gb_internal void lb_lower_bool_select(lbFastIselLowering *s, LLVMValueRef select) { LLVMValueRef c = LLVMGetOperand(select, 0); LLVMValueRef x = LLVMGetOperand(select, 1); @@ -1033,7 +1069,12 @@ gb_internal void lb_lower_for_fast_isel(lbModule *m) { } } for (LLVMValueRef store : work) { - if (!lb_scalarize_aggregate_store(&s, store)) { + if (lb_scalarize_aggregate_store(&s, store)) { + continue; + } + if (LLVMIsALoadInst(LLVMGetOperand(store, 0))) { + lb_lower_large_loaded_store(&s, store); + } else { lb_lower_large_constant_store(&s, store); } } From 6679c120f9792f5dec7334a565d1a8bfe7662031 Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 18:34:52 +0100 Subject: [PATCH 19/20] Wait for a polymorphic struct's fields before reading them in the subtype and comparison checks --- src/types.cpp | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/src/types.cpp b/src/types.cpp index 015b61ed3..11aba82bb 100644 --- a/src/types.cpp +++ b/src/types.cpp @@ -2068,6 +2068,7 @@ gb_internal bool is_type_map(Type *t) { } gb_internal void wait_for_union_variants(Type *t); +gb_internal void wait_for_struct_fields(Type *t); gb_internal bool is_type_union_maybe_pointer(Type *t) { t = base_type(t); @@ -2923,6 +2924,7 @@ gb_internal bool is_type_comparable(Type *t) { if (t->Struct.is_raw_union) { return is_type_simple_compare(t); } + wait_for_struct_fields(t); for_array(i, t->Struct.fields) { Entity *f = t->Struct.fields[i]; if (!is_type_comparable(f->type)) { @@ -2984,6 +2986,7 @@ gb_internal bool is_type_simple_compare(Type *t) { return is_type_simple_compare(t->Matrix.elem); case Type_Struct: + wait_for_struct_fields(t); if (t->Struct.is_simple) { return true; } @@ -3058,6 +3061,7 @@ gb_internal bool is_type_nearly_simple_compare(Type *t) { return is_type_nearly_simple_compare(t->Matrix.elem); case Type_Struct: + wait_for_struct_fields(t); if (t->Struct.is_simple) { return true; } @@ -3155,6 +3159,9 @@ gb_internal String lookup_subtype_polymorphic_field(Type *dst, Type *src) { // bool dst_is_ptr = dst != prev_dst; GB_ASSERT(is_type_struct(src) || is_type_union(src)); + if (src->kind == Type_Struct) { + wait_for_struct_fields(src); + } for_array(i, src->Struct.fields) { Entity *f = src->Struct.fields[i]; if (f->kind == Entity_Variable && f->flags & EntityFlags_IsSubtype) { @@ -3186,6 +3193,9 @@ gb_internal bool lookup_subtype_polymorphic_selection(Type *dst, Type *src, Sele // bool dst_is_ptr = dst != prev_dst; GB_ASSERT(is_type_struct(src) || is_type_union(src)); + if (src->kind == Type_Struct) { + wait_for_struct_fields(src); + } for_array(i, src->Struct.fields) { Entity *f = src->Struct.fields[i]; if (f->kind == Entity_Variable && f->flags & EntityFlags_IsSubtype) { @@ -5217,6 +5227,8 @@ gb_internal isize check_is_assignable_to_using_subtype(Type *src, Type *dst, isi if (!is_type_struct(src)) { return 0; } + // a polymorphic record is published for reuse before its fields are checked + wait_for_struct_fields(src); bool dst_is_polymorphic = is_type_polymorphic(dst); @@ -5259,6 +5271,7 @@ gb_internal bool check_is_assignable_to_using_offset_zero_subtype(Type *src, Typ if (!is_type_struct(src_struct)) { return false; } + wait_for_struct_fields(src_struct); // We check multiple fields in case of #raw_union, // but exit on the first field that is not at offset 0. From 448f859ceec07bdd37ff86dc2a1be4651b5f7f8f Mon Sep 17 00:00:00 2001 From: gingerBill Date: Sun, 4 Oct 2026 18:42:06 +0100 Subject: [PATCH 20/20] Read a switch's case values with `LLVMGetSwitchCaseValue` on LLVM 22 which no longer keeps them in its operands --- src/llvm_backend_opt.cpp | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/src/llvm_backend_opt.cpp b/src/llvm_backend_opt.cpp index d3502a147..f2f8d70c6 100644 --- a/src/llvm_backend_opt.cpp +++ b/src/llvm_backend_opt.cpp @@ -949,7 +949,7 @@ gb_internal void lb_truncate_bool_arguments(lbFastIselLowering *s, LLVMValueRef gb_internal void lb_lower_small_switch(lbFastIselLowering *s, LLVMValueRef sw) { enum {MAX_CASES = 3}; LLVMValueRef cond = LLVMGetOperand(sw, 0); - unsigned case_count = (cast(unsigned)LLVMGetNumOperands(sw) - 2) / 2; + unsigned case_count = LLVMGetNumSuccessors(sw) - 1; if (case_count == 0 || case_count > MAX_CASES || LLVMGetIntTypeWidth(LLVMTypeOf(cond)) > 64) { return; } @@ -979,7 +979,13 @@ gb_internal void lb_lower_small_switch(lbFastIselLowering *s, LLVMValueRef sw) { } } - LLVMValueRef cmp = LLVMBuildICmp(s->builder, LLVMIntEQ, cond, LLVMGetOperand(sw, 2 + 2*j), ""); +#if LLVM_VERSION_MAJOR >= 22 + // LLVM 22 keeps a switch's case values apart from its operands + LLVMValueRef case_value = LLVMGetSwitchCaseValue(sw, j+1); +#else + LLVMValueRef case_value = LLVMGetOperand(sw, 2 + 2*j); +#endif + LLVMValueRef cmp = LLVMBuildICmp(s->builder, LLVMIntEQ, cond, case_value, ""); LLVMBuildCondBr(s->builder, cmp, dest, else_block); from[j] = curr;