From 352a8c9c2592f8d8b88a8233a362a864b3972a94 Mon Sep 17 00:00:00 2001 From: Ed_ Date: Mon, 6 Jul 2026 10:10:02 -0400 Subject: [PATCH] Made the atom offset/label metaprogram! Still need to support more than one branch per-atom. --- code/duffle/gcc_asm.h | 4 +- code/duffle/gen/lottes_tape.offsets.h | 50 ++ code/duffle/gte.h | 2 +- code/duffle/lottes_tape.h | 36 +- code/duffle/mips.h | 37 - code/gte_hello/gen/hello_gte_tape.offsets.h | 18 + code/gte_hello/hello_gte.c | 5 + code/gte_hello/hello_gte_tape.c | 5 +- code/gte_hello/tape_atom.metadata.h | 39 ++ scripts/build_psyq.ps1 | 51 ++ scripts/tape_attom.offset_gen.meta.lua | 725 ++++++++++++++++++++ 11 files changed, 930 insertions(+), 42 deletions(-) create mode 100644 code/duffle/gen/lottes_tape.offsets.h create mode 100644 code/gte_hello/gen/hello_gte_tape.offsets.h create mode 100644 code/gte_hello/tape_atom.metadata.h create mode 100644 scripts/tape_attom.offset_gen.meta.lua diff --git a/code/duffle/gcc_asm.h b/code/duffle/gcc_asm.h index 763229b..6d0d41c 100644 --- a/code/duffle/gcc_asm.h +++ b/code/duffle/gcc_asm.h @@ -67,8 +67,8 @@ * * asm volatile("nop" : : : reg_str(R_RA), "memory"); // clobber list */ #define rlit_stringfy(n) "$" stringify(n) -#define rlit_tmpl(n) rlit_stringfy(tmpl(n,Code)) -#define rlit(n) rlit_tmpl(n) +#define rlit_tmpl(n) rlit_stringfy(tmpl(n,Code)) +#define rlit(n) rlit_tmpl(n) /* ------------------------------------------------------------------------ * * rgcc(n) — GCC-specific bundle for register-variable declarations. diff --git a/code/duffle/gen/lottes_tape.offsets.h b/code/duffle/gen/lottes_tape.offsets.h new file mode 100644 index 0000000..47d3b71 --- /dev/null +++ b/code/duffle/gen/lottes_tape.offsets.h @@ -0,0 +1,50 @@ +// Auto-generated by gen_atom_offsets.lua — DO NOT EDIT +// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h +#ifndef LOTTES_TAPE_OFFSETS_H +#define LOTTES_TAPE_OFFSETS_H + +#pragma region lottes_tape + +// Override the placeholder atom_offset() to dispatch via token pasting. +#undef atom_offset +#define atom_offset(name) atom_offset_##name + +// --- atom: sym (8 words) --- + + +// --- atom: tape_exit (2 words) --- + + +// --- atom: yield (4 words) --- + + +// --- atom: mips_flush_icache (13 words) --- + + +// --- atom: sync_prim_cursor (6 words) --- + + +// --- atom: set_gte_world (22 words) --- + + +// --- atom: rbind_cube_tri (6 words) --- + + +// --- atom: cube_tri (74 words) --- + + +// --- atom: rbind_floor_tri (6 words) --- + + +// --- atom: diag_yield (4 words) --- + + +// --- atom: diag_color (28 words) --- + + +// --- atom: diag_gte (34 words) --- + + +#pragma endregion lottes_tape + +#endif // LOTTES_TAPE_OFFSETS_H diff --git a/code/duffle/gte.h b/code/duffle/gte.h index 3224b2c..4c843ee 100644 --- a/code/duffle/gte.h +++ b/code/duffle/gte.h @@ -663,7 +663,7 @@ enum { asm_clobber: clb_system, rlit(R_T4), rlit(R_T5), rlit(R_T6) \ ) -#pragma region ASM DSL +#pragma endregion ASM DSL #pragma region Reserved diff --git a/code/duffle/lottes_tape.h b/code/duffle/lottes_tape.h index f3e3958..9ba1039 100644 --- a/code/duffle/lottes_tape.h +++ b/code/duffle/lottes_tape.h @@ -7,7 +7,10 @@ # include "memory.h" #endif +typedef U4 const MipsCode; +#define MipsAtom_(sym) MipsCode tmpl(code,sym) [] align_(4) = +#pragma region Tape Drive /* --------------------------------------------------------------------------- * TAPE DRIVE ABI & REGISTER ALIASES * --------------------------------------------------------------------------- @@ -63,12 +66,14 @@ FI_ TapeBuilder tb_make( FArena* arena) { return (TapeBuilder){ #define tb_emit_(tb, atom) tb_emit(tb, tmpl(code,atom)) FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; } -FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; } +FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; } FI_ Slice_U4 tb_end (TapeBuilder* tb) { tb_emit(tb,code_tape_exit); return (Slice_U4){ C_(U4*,tb->ptr), tb->used }; } FI_ Slice_U4 tb_slice(TapeBuilder tb) { return (Slice_U4){ C_(U4*,tb.ptr), tb.used }; } #define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,code_tape_exit)) +#pragma endregion Tape Drive + #pragma region Macro Mips Atom Components /* --------------------------------------------------------------------------- * MACRO ATOM Components (Reusable Assembly Components) @@ -150,6 +155,35 @@ FI_ void atombuilder_end(MipsAtomBuilder_R ab) { #pragma region Baked Mips Atoms // These atoms are resolved at compile time and are (usually) statically linked readonly data. +enum { + bios_flushcache = 0x44, + bios_table_addr = 0xA0, +}; + +/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0). + * + * Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack): + * 1. sp -= 8; sw $ra, 4($sp) ; save RA + * 2. $a0 = bios_flushcache (arg0) + * 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table + * 4. jalr $t0, $ra ; call BIOS(flushcache) + * nop ; branch delay slot + * 5. lw $ra, 4($sp); jr $ra ; restore & return + * 6. sp += 8 + */ +internal MipsAtom_(mips_flush_icache) { + add_ui(rstack_ptr, rstack_ptr, -8) /* sp -= 8 */ + , store_word(rret_addr, rstack_ptr, 4) /* sw $ra, 4($sp) */ + , add_ui(rret_0, rdiscard, bios_flushcache) /* addiu $a0, $0, 0x44 */ + , add_ui(rtmp_0, rdiscard, bios_table_addr) /* addiu $t0, $0, 0xA0 */ + , jump_link(rtmp_0, rret_addr) /* jalr $t0, $ra */ + , nop /* BD slot */ + , load_word(rret_addr, rstack_ptr, 4) /* lw $ra, 4($sp) */ + , jump_reg(rret_addr) /* jr $ra */ + , add_ui(rstack_ptr, rstack_ptr, 8) /* sp += 8 (BD) */ + , mac_yield() +}; + typedef Struct_(Binds_SyncPrimCursor) { U4 PrimtiveArena_Used; U4 PrimtiveBase; diff --git a/code/duffle/mips.h b/code/duffle/mips.h index 5bbc866..a9443c1 100644 --- a/code/duffle/mips.h +++ b/code/duffle/mips.h @@ -482,39 +482,6 @@ enum { _BitOffsets = 0 } \ } while (0 ) -// Binary Metaprogramming - -typedef U4 const MipsCode; -#define MipsAtom_(sym) MipsCode tmpl(code,sym) [] align_(4) = - -enum { - bios_flushcache = 0x44, - bios_table_addr = 0xA0, -}; - -/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0). - * - * Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack): - * 1. sp -= 8; sw $ra, 4($sp) ; save RA - * 2. $a0 = bios_flushcache (arg0) - * 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table - * 4. jalr $t0, $ra ; call BIOS(flushcache) - * nop ; branch delay slot - * 5. lw $ra, 4($sp); jr $ra ; restore & return - * 6. sp += 8 - */ -internal MipsAtom_(mips_flush_icache) { - add_ui(rstack_ptr, rstack_ptr, -8) /* sp -= 8 */ - , store_word(rret_addr, rstack_ptr, 4) /* sw $ra, 4($sp) */ - , add_ui(rret_0, rdiscard, bios_flushcache) /* addiu $a0, $0, 0x44 */ - , add_ui(rtmp_0, rdiscard, bios_table_addr) /* addiu $t0, $0, 0xA0 */ - , jump_link(rtmp_0, rret_addr) /* jalr $t0, $ra */ - , nop /* BD slot */ - , load_word(rret_addr, rstack_ptr, 4) /* lw $ra, 4($sp) */ - , jump_reg(rret_addr) /* jr $ra */ - , add_ui(rstack_ptr, rstack_ptr, 8) /* sp += 8 (BD) */ -}; -I_ void mips_flush_icache(void) { C_(VoidFn*, code_mips_flush_icache)(); } /* Standard clobber list for pure-MIPS asm volatile blocks: caller-saved * GPRs that the kernel treats as volatile (v0/v1/t0/t1/ra) plus the @@ -533,7 +500,3 @@ I_ void mips_flush_icache(void) { C_(VoidFn*, code_mips_flush_icache)(); } , jump_reg(rret_addr) \ , add_ui(rstack_ptr, rstack_ptr, 8) \ ) asm_clobber: clb_system ) - -void test_mips_asm() { - asm_mips_flush_icache(); -} diff --git a/code/gte_hello/gen/hello_gte_tape.offsets.h b/code/gte_hello/gen/hello_gte_tape.offsets.h new file mode 100644 index 0000000..fb1811c --- /dev/null +++ b/code/gte_hello/gen/hello_gte_tape.offsets.h @@ -0,0 +1,18 @@ +// Auto-generated by gen_atom_offsets.lua — DO NOT EDIT +// Source: C:\projects\Pikuma\ps1\code\gte_hello\hello_gte_tape.c +#ifndef HELLO_GTE_TAPE_OFFSETS_H +#define HELLO_GTE_TAPE_OFFSETS_H + +#pragma region hello_gte_tape + +// Override the placeholder atom_offset() to dispatch via token pasting. +#undef atom_offset +#define atom_offset(name) atom_offset_##name + +// --- atom: floor_tri (49 words) --- + +#define atom_offset_floor_tri_exit (17) + +#pragma endregion hello_gte_tape + +#endif // HELLO_GTE_TAPE_OFFSETS_H diff --git a/code/gte_hello/hello_gte.c b/code/gte_hello/hello_gte.c index 4c40bfe..882d744 100644 --- a/code/gte_hello/hello_gte.c +++ b/code/gte_hello/hello_gte.c @@ -12,7 +12,12 @@ #include "duffle/mips.h" #include "duffle/gp.h" #include "duffle/gte.h" + +# include "duffle/gen/lottes_tape.offsets.h" #include "duffle/lottes_tape.h" + +# include "tape_atom.metadata.h" +# include "gen/hello_gte_tape.offsets.h" #include "hello_gte.h" #include "hello_gte_tape.c" diff --git a/code/gte_hello/hello_gte_tape.c b/code/gte_hello/hello_gte_tape.c index b24d9cb..0b4b17a 100644 --- a/code/gte_hello/hello_gte_tape.c +++ b/code/gte_hello/hello_gte_tape.c @@ -1,6 +1,8 @@ #ifdef INTELLISENSE_DIRECTIVES # include "duffle/lottes_tape.h" # include "hello_gte.h" +# include "tape_atom.metadata.h" +# include "gen/hello_gte_tape.offsets.h" #endif #pragma region MACs @@ -32,7 +34,7 @@ internal MipsAtom_(floor_tri) { /* 4. Culling (Branch forward 29 instructions if Backface) */ gte_mf(R_T0, C2_MAC0), - nop, branch_le_zero(R_T0, 29), + nop, branch_le_zero(R_T0, atom_offset(floor_tri_exit)), nop, /* 5. Format Primitive */ @@ -54,6 +56,7 @@ internal MipsAtom_(floor_tri) { add_ui(R_PrimCur, R_PrimCur, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */ /* 9. Advance Input Cursor & Yield (Both branch targets land here) */ +atom_label(floor_tri_exit) add_ui(R_FaceCur, R_FaceCur, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */ mac_yield() }; diff --git a/code/gte_hello/tape_atom.metadata.h b/code/gte_hello/tape_atom.metadata.h new file mode 100644 index 0000000..ef89658 --- /dev/null +++ b/code/gte_hello/tape_atom.metadata.h @@ -0,0 +1,39 @@ +// tape_atom.metadata.h +// Single source of truth for instruction-word counts. +// Used by C (to define compile-time constants) AND Python (to count positions). +// +// Format: WORD_COUNT(MACRO_NAME, COUNT) +// One line per macro that appears in your atom sources. +// +// To regenerate: hand-count the instructions in each macro definition. +// (You'll only need to do this once per macro — they don't change often.) +#define WORD_COUNT(name, count) enum { words_##name = (count) }; + +WORD_COUNT(nop, 1) +WORD_COUNT(branch_le_zero, 1) +WORD_COUNT(branch_equal, 1) +WORD_COUNT(add_ui, 1) +WORD_COUNT(slt_u, 1) +WORD_COUNT(load_ui, 1) +WORD_COUNT(load_word, 1) +WORD_COUNT(load_half_u, 1) +WORD_COUNT(store_word, 1) +WORD_COUNT(gte_mf, 1) +WORD_COUNT(gte_mt, 1) +WORD_COUNT(gte_ct, 1) +WORD_COUNT(gte_sw, 1) +WORD_COUNT(gte_cmdw_rtpt, 1) +WORD_COUNT(gte_cmdw_nclip, 1) +WORD_COUNT(gte_avg_sort_z3, 1) +WORD_COUNT(mac_load_tri_indices, 3) +WORD_COUNT(mac_load_tri_verts, 18) +WORD_COUNT(mac_format_f3_color, 3) +WORD_COUNT(mac_gte_store_f3, 3) +WORD_COUNT(mac_insert_ot_tag, 11) +WORD_COUNT(mac_yield, 4) + +#undef WORD_COUNT + +// Used to define word markers for the lua metaprogram to calculate offsets from. +#define atom_label(sym) +// #define atom_offset(sym) // will be generated based on usage within a baked atom. diff --git a/scripts/build_psyq.ps1 b/scripts/build_psyq.ps1 index 53a4730..490f093 100644 --- a/scripts/build_psyq.ps1 +++ b/scripts/build_psyq.ps1 @@ -311,11 +311,60 @@ function build-graphis_hello { } # build-graphis_hello +function generate-TapeAtomOffsets {param( + [Parameter(Mandatory=$true)] + [string[]]$sources, + [Parameter(Mandatory=$true)] + [string]$metadata) + + $gen_atom_offsets_script = join-path $path_scripts 'tape_attom.offset_gen.meta.lua' + + $any_stale = $false + foreach ($src in $sources) { + $basename = [System.IO.Path]::GetFileNameWithoutExtension($src) + $dir = split-path -Path $src -Parent + $gen_dir = join-path $dir 'gen' + $out = join-path $gen_dir "$basename.offsets.h" + + if (-not (test-path $out)) { $any_stale = $true; break } + $src_mtime = (get-item $src).LastWriteTimeUtc + $out_mtime = (get-item $out).LastWriteTimeUtc + $meta_mtime = (get-item $metadata).LastWriteTimeUtc + if (($src_mtime -gt $out_mtime) -or ($meta_mtime -gt $out_mtime)) { + $any_stale = $true + break + } + } + + if (-not $any_stale) { + write-host "AtomOffs all $($sources.Count) source(s) up-to-date" -ForegroundColor DarkGray + return + } + + write-host "AtomOffs $($sources.Count) source(s)" -ForegroundColor Magenta + & lua $gen_atom_offsets_script $metadata @sources + if ($LASTEXITCODE -ne 0) { + write-error "Atom offset generation failed. Aborting." + exit 1 + } +} + function build-gte_hello { $includes += @() $path_module = join-path $path_code 'gte_hello' + $path_duffle = join-path $path_code 'duffle' + $path_gen = join-path $path_module 'gen' + $path_atom_metadata = join-path $path_module 'tape_atom.metadata.h' + + $atom_sources = @( + (join-path $path_duffle 'mips.h'), + (join-path $path_duffle 'lottes_tape.h'), + (join-path $path_module 'hello_gte_tape.c') + ) + generate-TapeAtomOffsets -sources $atom_sources -metadata $path_atom_metadata + $assemble_args = @() $assemble_args += $f_debug $assemble_args += $f_optimize_none @@ -353,6 +402,8 @@ function build-gte_hello { } build-gte_hello + +# NO idea if this works yet... function Send-ToEmulator { param( [string]$exePath ) diff --git a/scripts/tape_attom.offset_gen.meta.lua b/scripts/tape_attom.offset_gen.meta.lua new file mode 100644 index 0000000..d5e8ac3 --- /dev/null +++ b/scripts/tape_attom.offset_gen.meta.lua @@ -0,0 +1,725 @@ +#!/usr/bin/env lua +-- gen_atom_offsets.lua +-- +-- Finds every `MipsAtom_(name) { ... }` declaration in the given sources, +-- counts the words in each body using the WORD_COUNT manifest, computes +-- branch offsets for atom_label(name)/atom_offset(name) markers, and writes +-- one header per source into `/gen/.offsets.h`. +-- +-- Usage: +-- lua gen_atom_offsets.lua [source2 ...] + +-- ============================================================ +-- Character classification +-- ============================================================ + +local function is_space(c) + return c == " " or c == "\t" or c == "\n" or c == "\r" or c == "\v" or c == "\f" +end + +local function is_alpha(c) + if not c or #c == 0 then return false end + if c >= "a" and c <= "z" then return true end + if c >= "A" and c <= "Z" then return true end + return c == "_" +end + +local function is_digit(c) + return c and c >= "0" and c <= "9" +end + +local function is_alnum(c) + return is_alpha(c) or is_digit(c) +end + +-- ============================================================ +-- I/O +-- ============================================================ + +local function read_file(path) + local f = io.open(path, "r") + if not f then error("Cannot open " .. path) end + local content = f:read("*a") + f:close() + return content +end + +local function write_file(path, content) + local f = io.open(path, "w") + if not f then error("Cannot write " .. path) end + f:write(content) + f:close() +end + +local function ensure_dir(path) + os.execute('mkdir -p "' .. path .. '"') +end + +-- ============================================================ +-- String primitives (no patterns) +-- ============================================================ + +local function trim(s) + local a = 1 + while a <= #s and is_space(s:sub(a, a)) do a = a + 1 end + local b = #s + while b >= a and is_space(s:sub(b, b)) do b = b - 1 end + return s:sub(a, b) +end + +local function starts_with(s, prefix) + if #s < #prefix then return false end + for i = 1, #prefix do + if s:sub(i, i) ~= prefix:sub(i, i) then return false end + end + return true +end + +local function ends_with(s, suffix) + if #s < #suffix then return false end + local off = #s - #suffix + for i = 1, #suffix do + if s:sub(off + i, off + i) ~= suffix:sub(i, i) then return false end + end + return true +end + +local function find_byte(haystack, target, start) + for i = start or 1, #haystack do + if haystack:sub(i, i) == target then return i end + end + return nil +end + +local function dirname(path) + local last_sep = 0 + for i = 1, #path do + local c = path:sub(i, i) + if c == "/" or c == "\\" then last_sep = i end + end + if last_sep == 0 then return "." end + return path:sub(1, last_sep - 1) +end + +local function basename_no_ext(path) + local last_sep = 0 + for i = 1, #path do + local c = path:sub(i, i) + if c == "/" or c == "\\" then last_sep = i end + end + local a = last_sep + 1 + local last_dot = #path + 1 + for i = #path, a, -1 do + if path:sub(i, i) == "." then last_dot = i; break end + end + return path:sub(a, last_dot - 1) +end + +local function to_upper(s) + local out = "" + for i = 1, #s do + local code = string.byte(s, i) + if code >= 97 and code <= 122 then + out = out .. string.char(code - 32) + else + out = out .. s:sub(i, i) + end + end + return out +end + +local function to_alnum_underscore(s) + local out = "" + for i = 1, #s do + local c = s:sub(i, i) + if is_alnum(c) then out = out .. c + else out = out .. "_" end + end + return out +end + +local function pad_right(s, width) + while #s < width do s = s .. " " end + return s +end + +-- ============================================================ +-- Skip whitespace and comments +-- ============================================================ + +local function skip_ws_and_comments(source, i) + local len = #source + while i <= len do + local c = source:sub(i, i) + if is_space(c) then + i = i + 1 + elseif c == "/" and source:sub(i+1, i+1) == "/" then + while i <= len and source:sub(i, i) ~= "\n" do i = i + 1 end + elseif c == "/" and source:sub(i+1, i+1) == "*" then + i = i + 2 + while i <= len - 1 do + if source:sub(i, i) == "*" and source:sub(i+1, i+1) == "/" then + i = i + 2 + break + end + i = i + 1 + end + else + break + end + end + return i +end + +-- ============================================================ +-- Read identifier +-- ============================================================ + +local function read_ident(source, i) + if not is_alpha(source:sub(i, i)) then return nil, i end + local a = i + i = i + 1 + while i <= #source and is_alnum(source:sub(i, i)) do i = i + 1 end + return source:sub(a, i - 1), i +end + +-- ============================================================ +-- Read balanced (open_char, close_char) group, return inner + new pos +-- Skips strings and comments inside. +-- ============================================================ + +local function read_balanced(source, open_char, close_char, i) + if source:sub(i, i) ~= open_char then return nil, i end + i = i + 1 + local len = #source + local depth = 1 + local a = i + while i <= len and depth > 0 do + local c = source:sub(i, i) + if c == open_char then + depth = depth + 1 + i = i + 1 + elseif c == close_char then + depth = depth - 1 + if depth == 0 then break end + i = i + 1 + elseif c == '"' then + i = i + 1 + while i <= len do + if source:sub(i, i) == "\\" then i = i + 2 + elseif source:sub(i, i) == '"' then i = i + 1; break + else i = i + 1 end + end + elseif c == "'" then + i = i + 1 + while i <= len do + if source:sub(i, i) == "\\" then i = i + 2 + elseif source:sub(i, i) == "'" then i = i + 1; break + else i = i + 1 end + end + elseif c == "/" and source:sub(i+1, i+1) == "/" then + while i <= len and source:sub(i, i) ~= "\n" do i = i + 1 end + elseif c == "/" and source:sub(i+1, i+1) == "*" then + i = i + 2 + while i <= len - 1 do + if source:sub(i, i) == "*" and source:sub(i+1, i+1) == "/" then + i = i + 2 + break + end + i = i + 1 + end + else + i = i + 1 + end + end + return source:sub(a, i - 1), i + 1 +end + +local function read_parens(source, i) return read_balanced(source, "(", ")", i) end +local function read_braces(source, i) return read_balanced(source, "{", "}", i) end +local function read_brackets(source, i) return read_balanced(source, "[", "]", i) end + +-- ============================================================ +-- Scan forward from `start`, skipping balanced (), [], {}, strings, comments. +-- Returns position of first occurrence of `target` char at top level, or nil. +-- ============================================================ + +local function scan_to_char(source, target, start) + local len = #source + local i = start + while i <= len do + local c = source:sub(i, i) + if c == target then + return i + elseif c == "(" then + local _, after = read_parens(source, i); i = after + elseif c == "{" then + local _, after = read_braces(source, i); i = after + elseif c == "[" then + local _, after = read_brackets(source, i); i = after + elseif c == '"' then + i = i + 1 + while i <= len do + if source:sub(i, i) == "\\" then i = i + 2 + elseif source:sub(i, i) == '"' then i = i + 1; break + else i = i + 1 end + end + elseif c == "'" then + i = i + 1 + while i <= len do + if source:sub(i, i) == "\\" then i = i + 2 + elseif source:sub(i, i) == "'" then i = i + 1; break + else i = i + 1 end + end + elseif c == "/" and source:sub(i+1, i+1) == "/" then + while i <= len and source:sub(i, i) ~= "\n" do i = i + 1 end + elseif c == "/" and source:sub(i+1, i+1) == "*" then + i = i + 2 + while i <= len - 1 do + if source:sub(i, i) == "*" and source:sub(i+1, i+1) == "/" then + i = i + 2 + break + end + i = i + 1 + end + else + i = i + 1 + end + end + return nil +end + +-- ============================================================ +-- Load WORD_COUNT manifest from metadata.h +-- ============================================================ + +local function load_word_counts(metadata_path) + local counts = {} + local content = read_file(metadata_path) + local len = #content + local i = 1 + local prefix = "WORD_COUNT(" + while i <= len do + local nl = find_byte(content, "\n", i) + local line_end = nl or (len + 1) + local line = content:sub(i, line_end - 1) + local trimmed = trim(line) + + if starts_with(trimmed, prefix) and ends_with(trimmed, ")") then + local inner = trimmed:sub(#prefix + 1, #trimmed - 1) + local comma = find_byte(inner, ",", 1) + if comma then + local name = trim(inner:sub(1, comma - 1)) + local cnt = trim(inner:sub(comma + 1)) + counts[name] = tonumber(cnt) + end + end + + i = line_end + 1 + end + return counts +end + +-- ============================================================ +-- Count words for a single comma-separated token +-- ============================================================ + +local function word_count_of_token(token, word_counts) + local i = 1 + local len = #token + while i <= len and is_space(token:sub(i, i)) do i = i + 1 end + if i > len then return 0 end + local name, after = read_ident(token, i) + if not name then return 1 end + local j = skip_ws_and_comments(token, after) + if token:sub(j, j) == "(" then + local wc = word_counts[name] + if wc then return wc end + io.stderr:write(" warning: unknown macro '" .. name .. "', assuming 1 word\n") + return 1 + end + return 1 +end + +-- ============================================================ +-- Split brace-body into top-level comma-separated tokens +-- ============================================================ + +local function split_top_level_commas(body) + local tokens = {} + local len = #body + local i = 1 + local token_start = 1 + while i <= len do + local c = body:sub(i, i) + if c == "(" then + local _, after = read_parens(body, i); i = after + elseif c == "{" then + local _, after = read_braces(body, i); i = after + elseif c == "[" then + local _, after = read_brackets(body, i); i = after + elseif c == '"' then + i = i + 1 + while i <= len do + if body:sub(i, i) == "\\" then i = i + 2 + elseif body:sub(i, i) == '"' then i = i + 1; break + else i = i + 1 end + end + elseif c == "'" then + i = i + 1 + while i <= len do + if body:sub(i, i) == "\\" then i = i + 2 + elseif body:sub(i, i) == "'" then i = i + 1; break + else i = i + 1 end + end + elseif c == "/" and body:sub(i+1, i+1) == "/" then + while i <= len and body:sub(i, i) ~= "\n" do i = i + 1 end + elseif c == "/" and body:sub(i+1, i+1) == "*" then + i = i + 2 + while i <= len - 1 do + if body:sub(i, i) == "*" and body:sub(i+1, i+1) == "/" then + i = i + 2 + break + end + i = i + 1 + end + elseif c == "," then + table.insert(tokens, body:sub(token_start, i - 1)) + i = i + 1 + token_start = i + else + i = i + 1 + end + end + local last = body:sub(token_start, len) + if trim(last) ~= "" then + table.insert(tokens, last) + end + return tokens +end + +-- ============================================================ +-- Scan an atom body for atom_label/atom_offset markers, count words +-- ============================================================ + +local function scan_for_atom_markers(token, at_pos, labels, branches) + local i = 1 + local len = #token + while i <= len do + i = skip_ws_and_comments(token, i) + if i > len then break end + local c = token:sub(i, i) + if c == '"' then + -- Skip string literal + i = i + 1 + while i <= len do + if token:sub(i, i) == "\\" then i = i + 2 + elseif token:sub(i, i) == '"' then i = i + 1; break + else i = i + 1 end + end + elseif c == "'" then + -- Skip char literal + i = i + 1 + while i <= len do + if token:sub(i, i) == "\\" then i = i + 2 + elseif token:sub(i, i) == "'" then i = i + 1; break + else i = i + 1 end + end + elseif c == "/" and token:sub(i+1, i+1) == "/" then + while i <= len and token:sub(i, i) ~= "\n" do i = i + 1 end + elseif c == "/" and token:sub(i+1, i+1) == "*" then + i = i + 2 + while i <= len - 1 do + if token:sub(i, i) == "*" and token:sub(i+1, i+1) == "/" then + i = i + 2 + break + end + i = i + 1 + end + elseif is_alpha(c) then + local ident, after = read_ident(token, i) + if ident == "atom_label" or ident == "atom_offset" then + local arg_start = skip_ws_and_comments(token, after) + if token:sub(arg_start, arg_start) == "(" then + local inner, after_paren = read_parens(token, arg_start) + local n = 1 + while n <= #inner and is_space(inner:sub(n, n)) do n = n + 1 end + local ns = n + while n <= #inner and is_alnum(inner:sub(n, n)) do n = n + 1 end + local name = inner:sub(ns, n - 1) + if name ~= "" then + if ident == "atom_label" then + labels[name] = at_pos + else + table.insert(branches, {pos = at_pos, target = name}) + end + end + i = after_paren + else + i = arg_start + end + else + i = after + end + else + -- Anything else (parens, brackets, braces, commas, operators) — walk past + i = i + 1 + end + end +end + +local function scan_atom_body(body, word_counts) + local pos = 0 + local labels = {} + local branches = {} + + for _, tok in ipairs(split_top_level_commas(body)) do + local k = 1 + local tlen = #tok + while k <= tlen and is_space(tok:sub(k, k)) do k = k + 1 end + local leading_ident, leading_after = read_ident(tok, k) + + if leading_ident == "atom_label" or leading_ident == "atom_offset" then + scan_for_atom_markers(tok, pos, labels, branches) + else + local words = word_count_of_token(tok, word_counts) + scan_for_atom_markers(tok, pos, labels, branches) + pos = pos + words + end + end + + return labels, branches, pos +end + +-- ============================================================ +-- Token classification for atom detection +-- ============================================================ + +-- Skip past storage-class / qualifier noise. These appear before MipsCode +-- in raw expanded forms: `static`, `const`, the user's `internal`/`LP_`/ +-- `global` macros (which all expand to `static`), `RO_` (which expands to +-- a section attribute), plus standard C qualifiers. +local function skip_qualifiers(source, i) + local keywords = { + ["static"]=true, ["const"]=true, ["volatile"]=true, + ["extern"]=true, ["register"]=true, ["auto"]=true, + ["inline"]=true, ["typedef"]=true, + ["internal"]=true, ["LP_"]=true, ["global"]=true, ["gkknown"]=true + } + while true do + i = skip_ws_and_comments(source, i) + local ident, after = read_ident(source, i) + if not ident then return i end + if keywords[ident] then + i = after + else + return i + end + end +end + +-- Test whether `s` starts with literal `prefix` (no patterns). +local function has_prefix(s, prefix) + if #s < #prefix then return false end + for i = 1, #prefix do + if s:sub(i, i) ~= prefix:sub(i, i) then return false end + end + return true +end + +-- ============================================================ +-- Find every atom declaration in a source, both wrapped and raw +-- ============================================================ + +local function find_atoms(source_text) + local atoms = {} + local len = #source_text + local i = 1 + + -- First, scan inside source_text normally + local function try_wrapped_form(ident_pos) + local paren_pos = skip_ws_and_comments(source_text, ident_pos) + if source_text:sub(paren_pos, paren_pos) ~= "(" then + return nil -- not a MipsAtom_() call + end + local inner, after_paren = read_parens(source_text, paren_pos) + -- Extract name (first identifier from inner) + local n = 1 + while n <= #inner and is_space(inner:sub(n, n)) do n = n + 1 end + local ns = n + while n <= #inner and is_alnum(inner:sub(n, n)) do n = n + 1 end + local name = inner:sub(ns, n - 1) + if name == "" then return nil end + + local brace_pos = scan_to_char(source_text, "{", after_paren) + if not brace_pos then return nil end + local body, after_brace = read_braces(source_text, brace_pos) + return {name = name, body = body, after_brace = after_brace} + end + + local function try_raw_form(after_type_pos) + local next_pos = skip_ws_and_comments(source_text, after_type_pos) + local next_ident, next_after = read_ident(source_text, next_pos) + if not next_ident then return nil end + if not has_prefix(next_ident, "code_") then return nil end + if #next_ident <= 5 then return nil end -- bare "code_" — not an atom + + local atom_name = next_ident:sub(6) + local brace_pos = scan_to_char(source_text, "{", next_after) + if not brace_pos then return nil end + local body, after_brace = read_braces(source_text, brace_pos) + return {name = atom_name, body = body, after_brace = after_brace} + end + + while i <= len do + i = skip_ws_and_comments(source_text, i) + if i > len then break end + + -- Skip past storage-class noise + i = skip_qualifiers(source_text, i) + if i > len then break end + + local ident, after = read_ident(source_text, i) + if not ident then + i = i + 1 + elseif ident == "MipsAtom_" then + local atom = try_wrapped_form(after) + if atom then + table.insert(atoms, {name = atom.name, body = atom.body}) + i = atom.after_brace + else + i = i + 1 -- not actually MipsAtom_(), skip and continue + end + elseif ident == "MipsCode" then + local atom = try_raw_form(after) + if atom then + table.insert(atoms, {name = atom.name, body = atom.body}) + i = atom.after_brace + else + i = after -- some other MipsCode use; skip just this token + end + else + -- Anything else: skip just this identifier. The next loop + -- iteration will see whatever follows (might be more qualifiers, + -- another type keyword, etc.). + i = after + end + end + + return atoms +end + +-- ============================================================ +-- Compute branch offsets: target - branch - 1 +-- ============================================================ + +local function compute_offsets(labels, branches) + local results = {} + for _, br in ipairs(branches) do + local target = labels[br.target] + if not target then + error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.pos .. ")") + end + table.insert(results, {target = br.target, offset = target - br.pos - 1}) + end + return results +end + +-- ============================================================ +-- Generate header for one source +-- ============================================================ + +local function generate_header(source_path, atoms_data) + local basename = basename_no_ext(source_path) + local guard = to_alnum_underscore(to_upper(basename)) .. "_OFFSETS_H" + + local lines = {} + local function add(s) table.insert(lines, s) end + + add("// Auto-generated by gen_atom_offsets.lua — DO NOT EDIT") + add("// Source: " .. source_path) + add("#ifndef " .. guard) + add("#define " .. guard) + add("") + add("#pragma region " .. basename) + add("") + add("// Override the placeholder atom_offset() to dispatch via token pasting.") + add("#undef atom_offset") + add("#define atom_offset(name) atom_offset_##name") + add("") + + for _, atom in ipairs(atoms_data) do + add("// --- atom: " .. atom.name .. " (" .. atom.total_words .. " words) ---") + add("") + for _, r in ipairs(atom.offsets) do + local const_name = "atom_offset_" .. r.target + add("#define " .. pad_right(const_name, 40) .. " (" .. r.offset .. ")") + end + add("") + end + + add("#pragma endregion " .. basename) + add("") + add("#endif // " .. guard) + return table.concat(lines, "\n") .. "\n" +end + +-- ============================================================ +-- Process one source +-- ============================================================ + +local function process_source(source_path, word_counts) + local source = read_file(source_path) + local atoms_raw = find_atoms(source) + + if #atoms_raw == 0 then + io.stderr:write(" note: no MipsAtom_ declarations in " .. source_path .. "\n") + return + end + + local atoms_data = {} + for _, atom in ipairs(atoms_raw) do + local labels, branches, total = scan_atom_body(atom.body, word_counts) + local offsets = compute_offsets(labels, branches) + table.insert(atoms_data, { + name = atom.name, + total_words = total, + offsets = offsets + }) + end + + local basename = basename_no_ext(source_path) + local out_dir = dirname(source_path) .. "/gen" + ensure_dir(out_dir) + local out_path = out_dir .. "/" .. basename .. ".offsets.h" + write_file(out_path, generate_header(source_path, atoms_data)) + + local total_branches = 0 + for _, a in ipairs(atoms_data) do total_branches = total_branches + #a.offsets end + print(" " .. basename .. ": " .. #atoms_data .. " atom(s), " .. total_branches .. " branch(es)") + for _, a in ipairs(atoms_data) do + for _, r in ipairs(a.offsets) do + print(" " .. a.name .. " -> " .. r.target .. " : " .. r.offset) + end + end +end + +-- ============================================================ +-- Main +-- ============================================================ + +local function main(args) + if #args < 2 then + print("Usage: gen_atom_offsets.lua [source2 ...]") + os.exit(1) + end + + local metadata_path = args[1] + local sources = {} + for i = 2, #args do table.insert(sources, args[i]) end + + local word_counts = load_word_counts(metadata_path) + for _, src in ipairs(sources) do process_source(src, word_counts) end +end + +main({...})