Author SHA1 Message Date
ed a2d79d65eb amazing bug 2026-08-11 01:25:40 -04:00
ed bebcc6a585 wip: going to incremnetally test this. 2026-08-11 01:25:09 -04:00
ed ece21ed368 mark current crashing path. 2026-08-10 23:29:38 -04:00
ed 144c605ad8 some more review. not working still. 2026-08-10 23:04:43 -04:00
ed 4afd1af0fd started to review this... 2026-08-10 19:53:34 -04:00
ed 004a7eff19 WIP: not fully reviewed. Adds auto-register allocation + mips atom procs + wip resolve look at atoms + atom bundle... 2026-08-10 14:13:02 -04:00
ed e42c75a26a WIP: preparing for major changes to atoms to fullfill needs of resolve_look_at and atom ported normalize_v3s4. 2026-08-09 18:49:59 -04:00
ed 69f2c0d036 Prepping for: resolve_look_at impl. 2026-08-08 23:13:18 -04:00
ed b045856dd6 converted pad input for cam to mips atom 2026-08-08 18:23:28 -04:00
ed 68b87f1c8b Completed C-side of: Camera Transformation chapter. Now todo atom tape translation... 2026-08-08 16:42:07 -04:00
ed 917b764d95 pad_bios_init_start: annotate bios codes. 2026-08-08 13:32:25 -04:00
ed 773aa44013 reviewing pad input atoms further 2026-08-08 01:03:24 -04:00
ed 2b6fe53ce8 Stuff kept from hot-reload attempt 2026-08-06 10:41:45 -04:00
ed 01f7ceba7c buzzing brain. 2026-08-05 02:48:05 -04:00
ed 6f2eff920d some more review before bed. 2026-08-05 02:00:41 -04:00
ed f25765a7b7 Preparing for camera transformation chapter. 2026-08-05 01:21:25 -04:00
ed 2757aa4330 Fix bug with pad input processing (needed mac_yield load fallthrough case) 2026-08-05 01:08:01 -04:00
ed 748b58c5c5 Codebase overhaul. Metaprogram proofread (part 2). Starting to get serious.
Need to rewrite the ps1 lua metaprogram sometime soonish. Getting too bloated... need to consolidate code paths.

In this push codebase structure is starting to get a bit more realized. Decided todo now to match Pikuma's linking module files vods beginning to reorganize its codebase as well.
Atoms & atom components are not in their on *.atom.c files. (Not calling it tape.c as I don't really bake tapes like that outside of the unity c file so far...)

The lua metaprogram has had additional features added to it yet again to avoid hardcoding module handling and supporting multiple atom files per-module.
Either after the camera or cd-rom section I'll be most likely pausing to fully refactor the metaprogram. Possibly as a full re-write to get the loc minimal.
2026-08-04 23:34:00 -04:00
ed 6441dbc23e Proof-reading lua metaprogram (part 1) 2026-08-04 19:32:43 -04:00
ed 57fdb9e037 improvmenets to delay slot modeling (lua metaprogram) 2026-08-04 18:27:00 -04:00
ed b5953a723b add ac_yield_load and ac_yield_tail for delay slot optimization opportunities. 2026-08-04 17:25:02 -04:00
ed 888ffce859 Finished: Pikuma Linking multiple files (not applying to codebase only watched) 2026-08-04 16:49:02 -04:00
ed 7289e7c89c Added jump_rel (can't use abs jump with asm dsl). Fixes + improvements to ps1 asm meta passes. 2026-08-04 16:01:01 -04:00
ed 54a5bb9a31 starting to optimize 2026-08-04 12:59:51 -04:00
ed e0f4ac873d spamming load delay slots for now as a fix... 2026-08-04 09:07:53 -04:00
ed f17fa9165e wip: input was working... messed it up (bios snapshot reads) 2026-08-04 00:50:12 -04:00
ed 8282f8e902 overkill sio cruft, not keeping. 2026-08-03 10:12:06 -04:00
ed 9eb696ece8 drafting 2026-08-02 21:58:57 -04:00
ed 858e57f293 preparing to overhaul input handling 2026-08-02 17:49:24 -04:00
ed afcd9b86f0 Gaining clarity on tape abi.. screen_init atoms done. Time to finish rest of joypad course vods... 2026-08-02 15:19:52 -04:00
ed 43cd4e0344 WIP: working towards minimizing C-ABI & PsyQ CRT usage 2026-08-01 23:11:10 -04:00
ed 09dde54030 Finished(Controller Input): Reading Joypad State 2026-07-31 15:15:51 -04:00
ed 315e1b2c5e Fix(lua atom tape dsl): Bad-hardcode for source file line-table mapping in dwarf injection pass. 2026-07-31 14:28:50 -04:00
ed 02658d3609 Prepare for hello joypad! 2026-07-28 00:35:02 -04:00
ed dbc459b7e0 gte_hello -> hello_gte. gte is done, moving on to controller! 2026-07-28 00:17:23 -04:00
ed a704341fc6 Testing out the metaprogram with some optimization, need to remove some hardcoding later.. 2026-07-27 23:35:03 -04:00
ed 7421b32fd7 redundant nop reduction 2026-07-27 22:49:41 -04:00
ed e2eb74be19 Remove gte component result contracts (was a bad bodge in, for a later directive thats TODO) 2026-07-27 22:49:26 -04:00
ed 338f1fe46e Better reports from dsl metaprogram 2026-07-27 10:06:23 -04:00
65 changed files with 10147 additions and 3149 deletions
+4
View File
@@ -17,3 +17,7 @@ toolchain/PSn00bSDK
.vscode/settings.json .vscode/settings.json
toolchain/lfs toolchain/lfs
toolchain/lpeg toolchain/lpeg
scratch
toolchain/libpsn00b
scripts/pcsx_debug_helper.zip
+71 -36
View File
@@ -74,41 +74,7 @@
] ]
}, },
{ {
"name": "Debug: Hello GTE Psy-Q!", "name": "Debug: Hello GTE!",
"type": "gdb",
"request": "attach",
"target": "localhost:3333",
"remote": true,
"cwd": "${workspaceRoot}/build",
"valuesFormatting": "parseText",
"registerLimit": "1-32",
"frameFilters": false,
"showDevDebugOutput": false,
"printCalls": false,
"stopAtConnect": true,
"gdbpath": "gdb-multiarch",
"windows": {
"gdbpath": "gdb-multiarch.exe"
},
"osx": {
"gdbpath": "gdb"
},
"executable": "${workspaceRoot}/build/hello_gte.elf",
"setupCommands": [
{ "text": "set mi-async off" },
{ "text": "set remotetimeout 0" },
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
{ "text": "set logging redirect on" }
],
"autorun": [
"monitor reset shellhalt",
"load hello_gte.elf",
"tbreak main",
"continue"
]
},
{
"name": "Debug: Hello GTE Psy-Q! (atoms debug — DWARF-injected)",
"type": "gdb", "type": "gdb",
"request": "attach", "request": "attach",
"target": "localhost:3333", "target": "localhost:3333",
@@ -138,7 +104,76 @@
"monitor reset shellhalt", "monitor reset shellhalt",
"load build/hello_gte.dwarf-injected.elf", "load build/hello_gte.dwarf-injected.elf",
"source scripts/gdb/gdb_tape_atoms.gdb", "source scripts/gdb/gdb_tape_atoms.gdb",
"source build/gen/hello_gte.gdbinit", "tbreak main",
"continue"
]
},
{
"name": "Debug: Hello Joypad!",
"type": "gdb",
"request": "attach",
"target": "localhost:3333",
"remote": true,
"cwd": "${workspaceRoot}",
"valuesFormatting": "parseText",
"registerLimit": "1-32",
"frameFilters": false,
"showDevDebugOutput": false,
"printCalls": false,
"stopAtConnect": true,
"gdbpath": "gdb-multiarch",
"windows": {
"gdbpath": "gdb-multiarch.exe"
},
"osx": {
"gdbpath": "gdb"
},
"executable": "${workspaceRoot}/build/hello_joypad.dwarf-injected.elf",
"setupCommands": [
{ "text": "set mi-async off" },
{ "text": "set remotetimeout 0" },
{ "text": "set logging file build/gen/hello_joypad.gdb.log" },
{ "text": "set logging redirect on" }
],
"autorun": [
"monitor reset shellhalt",
"load build/hello_joypad.dwarf-injected.elf",
"source scripts/gdb/gdb_tape_atoms.gdb",
"tbreak main",
"continue"
]
},
{
"name": "Debug: Hello Camera!",
"type": "gdb",
"request": "attach",
"target": "localhost:3333",
"remote": true,
"cwd": "${workspaceRoot}",
"valuesFormatting": "parseText",
"registerLimit": "1-32",
"frameFilters": false,
"showDevDebugOutput": false,
"printCalls": false,
"stopAtConnect": true,
"gdbpath": "gdb-multiarch",
"windows": {
"gdbpath": "gdb-multiarch.exe"
},
"osx": {
"gdbpath": "gdb"
},
"executable": "${workspaceRoot}/build/hello_camera.dwarf-injected.elf",
"setupCommands": [
{ "text": "set mi-async off" },
{ "text": "set remotetimeout 0" },
{ "text": "set logging file build/gen/hello_camera.gdb.log" },
{ "text": "set logging redirect on" }
],
"autorun": [
"monitor reset shellhalt",
"load build/hello_camera.dwarf-injected.elf",
"source scripts/gdb/gdb_tape_atoms.gdb",
"tbreak main", "tbreak main",
"continue" "continue"
] ]
+14
View File
@@ -0,0 +1,14 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
#endif
enum {
bios_init_pad_2 = 0x12,
bios_start_pad_2 = 0x13,
bios_flushcache = 0x44,
bios_table_addr = 0xA0,
bios_btable_addr = 0xB0,
};
enum {
bios_pad_buffer_size = 0x22,
};
@@ -1,5 +1,5 @@
/* /*
* atom_dsl.h * dsl.atom.h
* ============================================================================ * ============================================================================
* *
* ATOM DSL: Annotation layer for tape atoms (lottes_tape.h). * ATOM DSL: Annotation layer for tape atoms (lottes_tape.h).
@@ -57,7 +57,6 @@
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
#pragma once #pragma once
// #include <stdint.h>
#endif #endif
/* ============================================================================ /* ============================================================================
@@ -76,6 +75,26 @@
* ----------------------------------------------------------------------------*/ * ----------------------------------------------------------------------------*/
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */ #define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
// ----------------------------------------------------------------------------
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
// enum {
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
// };
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
// ----------------------------------------------------------------------------
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
// enum {
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
// phase_auto_reg(cube_g4, R_Temp1),
// };
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
/* ============================================================================ /* ============================================================================
* atom_info : * atom_info :
* MipsAtom_(cube_tri) atom_info( * MipsAtom_(cube_tri) atom_info(
@@ -148,12 +167,12 @@
* ... body ... * ... body ...
* atom_label(bounds_chk) another anchor * atom_label(bounds_chk) another anchor
* *
* atom_offset(culling, bounds_chk) resolved by gen/.offsets.h * atom_offset(culling, bounds_chk) resolved by gen/offsets.h
* *
* The metaprogram generates gen/atom_offsets.h with one #define with the offset value per atom_offset(F, T) call. * The metaprogram generates gen/offsets.h with one #define with the offset value per atom_offset(F, T) call.
* The preprocessor then expands the call to the right immediate value. * The preprocessor then expands the call to the right immediate value.
* *
* If gen/atom_offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails. * If gen/offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails.
* ============================================================================*/ * ============================================================================*/
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T #define atom_offset(F, T) atom_offset_ ## F ## _ ## T
// atom_label is a pure annotation for the metaprogram's offset calculations. // atom_label is a pure annotation for the metaprogram's offset calculations.
+23 -15
View File
@@ -28,8 +28,9 @@
#define internal static // internal #define internal static // internal
#define asm __asm__ #define asm __asm__
#define align_(value) __attribute__((aligned (value))) // for easy alignment
#define A_(data) (& data)
#define align_(value) __attribute__((aligned (value))) // for easy alignment
#define align_(value) __attribute__((aligned (value))) // for easy alignment #define align_(value) __attribute__((aligned (value))) // for easy alignment
#define C_(type,data) ((type)(data)) // for enforced precedence #define C_(type,data) ((type)(data)) // for enforced precedence
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path #define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
@@ -43,7 +44,9 @@
#define R_ restrict #define R_ restrict
#define V_ volatile #define V_ volatile
// Fictional, used for intiution.
#pragma region Fictional //, used for intiution
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers) #define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation, #define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
// unlocking the compilers ability to safely pack data across multiple parallel SIMD lanes (vectorization). // unlocking the compilers ability to safely pack data across multiple parallel SIMD lanes (vectorization).
@@ -67,7 +70,8 @@
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_) #define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_) #define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_) #define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
//end of: Fictional.
#pragma endreigon Fictional
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping. // R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
@@ -130,21 +134,22 @@ typedef __UINT32_TYPE__ TSet_(B4);
#define u4_v(value) C_(U4 V_*, value) #define u4_v(value) C_(U4 V_*, value)
enum { false = 0, true = 1, true_overflow, }; enum { false = 0, true = 1, true_overflow, };
#define u4_lo(value) ((value) & 0xFFFFU) #define u4_lo(value) (u4_(value) & 0xFFFFU)
#define u4_hi(value) ((value) >> 12) #define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
typedef void Proc_(VoidFn) (void); typedef void Proc_(VoidFn) (void);
#define kilo(n) (C_(U4, n) << 10) #define kilo(n) (C_(U4, n) << 10)
#define mega(n) (C_(U4, n) << 20) #define mega(n) (C_(U4, n) << 20)
#define giga(n) (C_(U4, n) << 30) #define giga(n) (C_(U4, n) << 30)
#define tera(n) (C_(U4, n) << 40) #define tera(n) (C_(U4, n) << 40)
#define null C_(U4, 0)
#define nullptr C_(void*, 0) #define null C_(U4, 0)
#define O_(type, field) (C_(U4, & C_(type*,0)->field)) #define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OT_(field) O_(typeof_ptr(& field), filed)) #define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
#define S_(data) C_(U4, sizeof(data)) #define OT_(field) O_(typeof_ptr(& field), filed))
#define S_(data) C_(U4, sizeof(data))
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b)) #define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b)) #define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
@@ -164,6 +169,8 @@ def_signed_ops(le, <=)
#undef def_signed_ops #undef def_signed_ops
#undef def_signed_op #undef def_signed_op
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
#if 0
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__) #define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
#define add_s(a,b) def_generic_sop(add,a,b) #define add_s(a,b) def_generic_sop(add,a,b)
#define sub_s(a,b) def_generic_sop(sub,a,b) #define sub_s(a,b) def_generic_sop(sub,a,b)
@@ -173,6 +180,7 @@ def_signed_ops(le, <=)
#define ge_s(a,b) def_generic_sop(ge, a,b) #define ge_s(a,b) def_generic_sop(ge, a,b)
#define le_s(a,b) def_generic_sop(le, a,b) #define le_s(a,b) def_generic_sop(le, a,b)
#undef def_generic_sop #undef def_generic_sop
#endif
#define alignas _Alignas #define alignas _Alignas
#define alignof _Alignof #define alignof _Alignof
+15 -23
View File
@@ -50,17 +50,13 @@
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__)) #define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines // Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
/* reg_str(n) — Stringify an integer register id into the GCC asm /* reg_str(n) — Stringify an integer register id into the GCC asm string form (e.g. 12 → "$12").
* string form (e.g. 12 → "$12"). Use this anywhere GCC's parser * Use this anywhere GCC's parser expects a literal string identifying a register: clobber lists,
* expects a literal string identifying a register: clobber lists, * asm templates, etc. The two-level macro is the standard preprocessor idiom for forcing one level of expansion before stringify —
* asm templates, etc. The two-level macro is the standard preprocessor * without it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead of expanding `R_T4` to its value first.
* idiom for forcing one level of expansion before stringify — without
* it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead
* of expanding `R_T4` to its value first.
* *
* For declaring a register variable bound to a specific GPR, use the * For declaring a register variable bound to a specific GPR, use the `rgcc(n)` bundle from gcc_asm.h instead —
* `rgcc(n)` bundle from gcc_asm.h instead — it adds the `__asm__()` * it adds the `__asm__()` qualifier around the string.
* qualifier around the string.
* *
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose * register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled * register V3_S2* p0 rgcc(R_T4) = ...; // bundled
@@ -85,21 +81,19 @@
* - The string "$12" is derived from it via reg_str, so they cannot drift apart. * - The string "$12" is derived from it via reg_str, so they cannot drift apart.
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise. * - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
* *
* tmpl defined in dsl.h (the token-paste glue). * tmpl defined in dsl.h (token-paste glue).
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific. * rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
* Anyone porting to a different compiler's asm dialect overrides rgcc, * Anyone porting to a different compiler's asm dialect overrides rgcc,
* and the integer→string derivation in rlit can be retargeted in one place. * and the integer→string derivation in rlit can be retargeted in one place.
* *
* For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`. * For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`.
* ------------------------------------------------------------------------ */ * ------------------------------------------------------------------------ */
#define rgcc(n) __asm__(rlit(n)) #define rgcc(n) __asm__(rlit(n))
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used /* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used by the placeholder-pun macros
* by the placeholder-pun macros (the .word bodies are fully baked * (the .word bodies are fully baked at compile time and have no runtime operand references),
* at compile time and have no runtime operand references), but kept * but kept here for completeness in case a future asm template needs to refer to a runtime input by position.
* here for completeness in case a future asm template needs to refer * Mirror of rgcc but produces "%N" instead of "$N". */
* to a runtime input by position. Mirror of rgcc but produces "%N"
* instead of "$N". */
#define rgcc_ref_(n) "%" #n #define rgcc_ref_(n) "%" #n
#define rgcc_ref(n) rgcc_ref_(n) #define rgcc_ref(n) rgcc_ref_(n)
@@ -147,11 +141,9 @@
9, 8, 7, 6, 5, 4, 3, 2, 1, 0)) 9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
/* --- 2. String Concatenation Helpers --- * /* --- 2. String Concatenation Helpers --- *
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's * NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's asm-parser rejects `%cN` in this position with "invalid use of '%c'".
* asm-parser rejects `%cN` in this position with "invalid use of '%c'". * The `%cN` form is for printing *character* constants; for arbitrary integer immediates (the only kind `"i"(...)` produces),
* The `%cN` form is for printing *character* constants; for arbitrary * the plain `%N` form is the right one. Both expand to the bare immediate.
* integer immediates (the only kind `"i"(...)` produces), the plain
* `%N` form is the right one. Both expand to the bare immediate.
*/ */
#define GCC_ASM_W1 "%0" #define GCC_ASM_W1 "%0"
#define GCC_ASM_W2 GCC_ASM_W1 ", %1" #define GCC_ASM_W2 GCC_ASM_W1 ", %1"
-139
View File
@@ -1,139 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua — DO NOT EDIT
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
#ifndef WORD_COUNT
#define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif
/* atom_dbg_skip */
/* ---------------------------------------------------------------------------
* MACRO ATOM Components (Reusable Assembly Components)
* These do NOT yield. They are expanded inline inside Tape Atoms.
* ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield).
#define mac_yield(...) \
load_word(R_AtomJmp, R_TapePtr, 0) \
, add_ui_self( R_TapePtr, S_(MipsCode)) \
, jump_reg( R_AtomJmp) \
, nop
WORD_COUNT(mac_yield, 4)
/* atom_dbg_skip */
/* Words: 3; Loads 3 S2 indices from the face array */
#define mac_load_tri_indices(...) \
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \
, load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)) \
, load_half_u(R_T2, R_FaceCursor, 2 * S_(S2))
WORD_COUNT(mac_load_tri_indices, 3)
/* atom_dbg_skip */
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
#define mac_gte_load_tri_verts(...) \
shift_lleft(R_AT, R_T0, v3s2_byteoff) \
, add_u_self(R_AT, R_VertBase) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY0) \
, gte_mv_to_data_r(R_V1, C2_VZ0) \
, shift_lleft(R_AT, R_T1, v3s2_byteoff) \
, add_u_self(R_AT, R_VertBase) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY1) \
, gte_mv_to_data_r(R_V1, C2_VZ1) \
, shift_lleft(R_AT, R_T2, v3s2_byteoff) \
, add_u_self(R_AT, R_VertBase) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY2) \
, gte_mv_to_data_r(R_V1, C2_VZ2)
WORD_COUNT(mac_gte_load_tri_verts, 18)
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
#define mac_insert_ot_tag_f3(...) \
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, R_PrimCursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
, shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag_f3, 11)
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
#define mac_insert_ot_tag_g4(...) \
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, R_PrimCursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
, shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag_g4, 11)
/* atom_dbg_skip */
#define mac_pack_color_word(off, cmd, r, g, b) \
load_upper_i(R_AT, (cmd) << 8 | (b)) \
, or_i_self( R_AT, ((g) << 8) | (r)) \
, store_word( R_AT, R_PrimCursor, (off))
WORD_COUNT(mac_pack_color_word, 3)
/* atom_dbg_skip */
#define mac_format_f3_color(r, g, b) \
mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
#define mac_gte_store_f3_post_rtpt(...) \
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
WORD_COUNT(mac_gte_store_f3_post_rtpt, 3)
#define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
, mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1) \
, mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2) \
, mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3)
WORD_COUNT(mac_format_g4_color, 12)
/* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
* three registers aligned with v0/v1/v2 you must store before RTPS).
* The macro name declares the pipeline position; check #6 (GTE state-
* machine validation) verifies the call site matches the declaration. */
#define mac_gte_store_g4_p012_post_rtpt_pre_rtps(...) \
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3)
/* atom_dbg_skip */
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
* prevents).
*/
#define mac_gte_store_g4_p3_post_rtps(...) \
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3))
WORD_COUNT(mac_gte_store_g4_p3_post_rtps, 1)
-9
View File
@@ -1,9 +0,0 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
#pragma once
#pragma region lottes_tape
#pragma endregion lottes_tape
+247
View File
@@ -0,0 +1,247 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\duffle/
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
#ifndef WORD_COUNT
#define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif
/* atom_dbg_skip */
/* ---------------------------------------------------------------------------
* MACRO ATOM Components (Reusable Assembly Components)
* These do NOT yield. They are expanded inline inside Tape Atoms.
* ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield).
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
// - mac_yield_load() + mac_yield_tail():
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
#define mac_yield(...) \
load_word(R_AtomJmp, R_TapePtr, 0) \
, add_ui_self( R_TapePtr, S_(MipsCode)) \
, jump_reg( R_AtomJmp) \
, nop
WORD_COUNT(mac_yield, 4)
/* atom_dbg_skip */
#define mac_yield_load(...) \
load_word(R_AtomJmp, R_TapePtr, 0)
WORD_COUNT(mac_yield_load, 1)
/* atom_dbg_skip */
#define mac_yield_tail(...) \
add_ui_self(R_TapePtr, S_(MipsCode)) \
, jump_reg( R_AtomJmp) \
, nop
WORD_COUNT(mac_yield_tail, 3)
/* atom_dbg_skip */
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
load_half( rs_x, r_base, O_(V3_S2,x)) \
, load_half( rs_y, r_base, O_(V3_S2,y))
WORD_COUNT(mac_load_v2s2, 2)
/* atom_dbg_skip */
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
store_half(rt_x, base, offset + O_(V2_S2,x)) \
, store_half(rt_y, base, offset + O_(V2_S2,y))
WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
load_word( rs_x, r_base, O_(V3_S4,x)) \
, load_word( rs_y, r_base, O_(V3_S4,y)) \
, load_word( rs_z, r_base, O_(V3_S4,z))
WORD_COUNT(mac_load_v3s4, 3)
/* atom_dbg_skip */
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
store_word(rt_x, base, offset + O_(V3_S4,x)) \
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
, store_word(rt_z, base, offset + O_(V3_S4,z))
WORD_COUNT(mac_store_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
sub_s(rds_x, rds_x, rt_x) \
, sub_s(rds_y, rds_y, rt_y) \
, sub_s(rds_z, rds_z, rt_z)
WORD_COUNT(mac_sub_v3s4, 3)
/* atom_dbg_skip */
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
, store_half(rt_height, base, offset + O_(Rect_S2,height))
WORD_COUNT(mac_store_rects2, 4)
/* atom_dbg_skip */
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
, load_half_u(r_i1, r_face_cusor, 1 * S_(S2)) \
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
WORD_COUNT(mac_load_tri_indices, 3)
/* atom_dbg_skip */
#define mac_gte_store_f3(r_primitive_cursor) \
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)) \
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2))
WORD_COUNT(mac_gte_store_f3, 3)
/* atom_dbg_skip */
#define mac_gte_load_tri_verts(r_vert_base, r_v0, r_v1, r_v2) \
shift_lleft(R_AT, r_v0, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY0) \
, gte_mv_to_data_r(R_V1, C2_VZ0) \
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY1) \
, gte_mv_to_data_r(R_V1, C2_VZ1) \
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY2) \
, gte_mv_to_data_r(R_V1, C2_VZ2)
WORD_COUNT(mac_gte_load_tri_verts, 18)
/* atom_dbg_skip */
#define mac_gte_store_g4_p012(r_primitive_cursor) \
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)) \
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)) \
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2))
WORD_COUNT(mac_gte_store_g4_p012, 3)
/* atom_dbg_skip */
#define mac_gte_store_g4_p3(r_primitive_cursor) \
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
WORD_COUNT(mac_gte_store_g4_p3, 1)
/* atom_dbg_skip */
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop \
, gte_cmdw_sqr \
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
WORD_COUNT(mac_gte_sqr_v3, 8)
/* atom_dbg_skip */
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_dx, C2_MAC1) \
, gte_mv_from_data_r(r_dy, C2_MAC2) \
, gte_mv_from_data_r(r_dz, C2_MAC3) \
, shift_aright_var(r_dx, r_dx, r_shift) \
, shift_aright_var(r_dy, r_dy, r_shift) \
, shift_aright_var(r_dz, r_dz, r_shift)
WORD_COUNT(mac_gte_gpf_scale, 13)
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
load_upper_i(reg_transfer, cmd >> 16) \
, or_i_self( reg_transfer, cmd & 0xFFFF) \
, store_word( reg_transfer, reg_base, port)
WORD_COUNT(mac_gcmd_push, 3)
/* atom_dbg_skip */
#define mac_store_rgb8(rr, rg, rb, base, offset) \
store_byte(rr, base, offset + O_(RGB8,r)) \
, store_byte(rg, base, offset + O_(RGB8,g)) \
, store_byte(rb, base, offset + O_(RGB8,b))
WORD_COUNT(mac_store_rgb8, 3)
/* atom_dbg_skip */
#define mac_pack_color_word(r_base, off, cmd, r, g, b) \
load_upper_i(R_AT, (cmd) << 8 | (b)) \
, or_i_self( R_AT, ((g) << 8) | (r)) \
, store_word( R_AT, r_base, (off))
WORD_COUNT(mac_pack_color_word, 3)
/* atom_dbg_skip */
#define mac_format_f3_color(r_base, r, g, b) \
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3)
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2) \
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
WORD_COUNT(mac_format_g4_color, 12)
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag, 11)
/* atom_dbg_skip */
#define mac_pad_set_centered_axes(r_state, r_scratch) \
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF) \
, or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF) \
, store_word( r_scratch, r_state, O_(PadState,axes))
WORD_COUNT(mac_pad_set_centered_axes, 3)
/* atom_dbg_skip */
#define mac_pad_set_id_byte(r_state, r_id, id_value) \
add_ui( r_id, R_0, id_value) \
, store_byte(r_id, r_state, O_(PadState,id))
WORD_COUNT(mac_pad_set_id_byte, 2)
/* atom_dbg_skip */
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
add_ui( r_tmp, R_0, pad_status) \
, store_word(r_tmp, r_state, O_(PadState,status))
WORD_COUNT(mac_pad_set_status, 2)
/* atom_dbg_skip */
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
nor_u( r_buttons, r_buttons, R_0) \
, store_half( r_buttons, r_pad_state, O_(PadState, buttons))
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
+65
View File
@@ -0,0 +1,65 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\duffle\
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
#pragma once
#pragma region duffle
// --- atom: normalize_v3s4 (62 words) ---
#define _atom_offset_srav_path_aligned_done 6
#define _atom_offset_aligned_done_srav_path 1
enum {
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
};
// --- atom: pad_bios_snapshot (84 words) ---
#define _atom_offset_snap_root_skip_disconnected 10
#define _atom_offset_disconnected_snap_end 65
#define _atom_offset_case_2_id_dispatch 9
#define _atom_offset_pending_snap_end 54
#define _atom_offset_id_dispatch_try_analog_stick 12
#define _atom_offset_id_dispatch_snap_end 40
#define _atom_offset_try_analog_stick_try_analog_pad 13
#define _atom_offset_analog_stick_snap_end 25
#define _atom_offset_try_analog_pad_try_unsupported 12
#define _atom_offset_analog_pad_snap_end 10
enum {
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
};
#pragma endregion duffle
+60
View File
@@ -0,0 +1,60 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "dsl.h"
# include "gp.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_gcmd_push(MipsAtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_gcmd_push, ab, {
load_upper_i(reg_transfer, cmd >> 16),
or_i_self( reg_transfer, cmd & 0xFFFF),
store_word( reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_store_rgb8(MipsAtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, {
store_byte(rr, base, offset + O_(RGB8,r)),
store_byte(rg, base, offset + O_(RGB8,g)),
store_byte(rb, base, offset + O_(RGB8,b)),
})
FI_ Slice_MipsCode ac_pack_color_word(MipsAtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, {
load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, r_base, (off)),
})
FI_ Slice_MipsCode ac_format_f3_color(MipsAtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
FI_ Slice_MipsCode ac_format_g4_color(MipsAtomBuilder_R ab, U4 r_prim_cursor,
U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1,
U1 r2, U1 g2, U1 b2,
U1 r3, U1 g3, U1 b3)
MipsAtomComp_Proc_(ac_format_g4_color, ab, {
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
})
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
I_ Slice_MipsCode ac_insert_ot_tag(MipsAtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
})
#pragma endregion MACs (Mips Atom Components)
+291 -243
View File
@@ -39,8 +39,8 @@
/* ============================================================================ /* ============================================================================
* Hardware MMIO Addresses * Hardware MMIO Addresses
* ============================================================================ * ============================================================================
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2 * PSX GPU has two 32-bit ports in the I/O register region at KSEG2 0x1F800000+.
* 0x1F800000+. GP0 (offset 0x10) is the data port (commands + params). * GP0 (offset 0x10) is the data port (commands + params).
* GP1 (offset 0x14) is the control port (status, ctrl writes). * GP1 (offset 0x14) is the control port (status, ctrl writes).
* ============================================================================ */ * ============================================================================ */
/* IO base address (KSEG2 0x1F800000+ for the I/O register region). /* IO base address (KSEG2 0x1F800000+ for the I/O register region).
@@ -49,18 +49,18 @@
* `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word). * `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word).
* Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */ * Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */
enum { enum {
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */ IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */ IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros /* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
* that pin a register to IO_BASE_ADDR and access ports via offsets: * that pin a register to IO_BASE_ADDR and access ports via offsets:
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0 * sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */ * sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
GPIO_PORT0_OFFSET = 0x1810, GPIO_PORT0_OFFSET = 0x1810,
GPIO_PORT1_OFFSET = 0x1814, GPIO_PORT1_OFFSET = 0x1814,
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET, HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET, HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
}; };
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR) #define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
@@ -73,66 +73,64 @@ enum {
* GP0 command byte constants + Layer 1 (GPU bitfield shifts) * GP0 command byte constants + Layer 1 (GPU bitfield shifts)
* ============================================================================ * ============================================================================
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only. * 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
* The layer-1 bitfield-layout constants live in the same enum block so the encoder can reference them by name.
* NO macro body past this point uses a raw shift or raw mask. * NO macro body past this point uses a raw shift or raw mask.
* Every shift/width/mask is named here, named once.
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h. * Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp0_cmd_Nop = 0x00, gp0_cmd_Nop = 0x00,
/* Cache management */ /* Cache management */
gp0_cmd_ClearCache = 0x01, gp0_cmd_ClearCache = 0x01,
gp0_cmd_FillVram = 0x02, gp0_cmd_FillVram = 0x02,
gp0_cmd_CopyVram = 0x80, gp0_cmd_CopyVram = 0x80,
gp0_cmd_CopyVramChained = 0x81, gp0_cmd_CopyVramChained = 0x81,
gp0_cmd_ReadVram = 0xC0, gp0_cmd_ReadVram = 0xC0,
/* Polygons */ /* Polygons */
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */ gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */ gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */ gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */ gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */ gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */ gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */ gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */ gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
/* Lines */ /* Lines */
gp0_cmd_line_f2 = 0x40, gp0_cmd_line_f2 = 0x40,
gp0_cmd_line_g2 = 0x50, gp0_cmd_line_g2 = 0x50,
/* Sprites + Tiles + Rects */ /* Sprites + Tiles + Rects */
gp0_cmd_sprt_1 = 0x64, gp0_cmd_sprt_1 = 0x64,
gp0_cmd_sprt_8 = 0x74, gp0_cmd_sprt_8 = 0x74,
gp0_cmd_sprt_16 = 0x7C, gp0_cmd_sprt_16 = 0x7C,
gp0_cmd_tile_1 = 0x60, gp0_cmd_tile_1 = 0x60,
gp0_cmd_tile_8 = 0x68, gp0_cmd_tile_8 = 0x68,
gp0_cmd_tile_16 = 0x70, gp0_cmd_tile_16 = 0x70,
/* State setters (not drawing primitives; set render context). */ /* State setters (not drawing primitives; set render context). */
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */ gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
gp0_cmd_SetTextureWindow = 0xE2, gp0_cmd_SetTextureWindow = 0xE2,
gp0_cmd_SetDrawArea_TopLeft = 0xE3, gp0_cmd_SetDrawArea_TopLeft = 0xE3,
gp0_cmd_SetDrawArea_BotRight = 0xE4, gp0_cmd_SetDrawArea_BotRight = 0xE4,
gp0_cmd_SetDrawOffset = 0xE5, gp0_cmd_SetDrawOffset = 0xE5,
gp0_cmd_SetMaskBit = 0xE6, gp0_cmd_SetMaskBit = 0xE6,
/* bitfield shifts / widths / masks ---- /* bitfield shifts / widths / masks ----
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */ * Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
gp0_cmd_shift = 24, gp0_cmd_shift = 24,
gp0_cmd_width = 8, gp0_cmd_width = 8,
gp0_cmd_mask = 0xFF, gp0_cmd_mask = 0xFF,
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.): /* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
* bits 31..24 = command byte * bits 31..24 = command byte
* bits 23..16 = BLUE * bits 23..16 = BLUE
* bits 15..08 = GREEN * bits 15..08 = GREEN
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */ * bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF, gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF, gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF, gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF, gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
}; };
/* ============================================================================ /* ============================================================================
@@ -171,10 +169,13 @@ enum {
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b)) #define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
/* Cache management — bare-cmd words (no color/range payload). */ /* Cache management — bare-cmd words (no color/range payload). */
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache) #define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram) #define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram) #define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram) #define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
/* NOP — bare-cmd word (no effect; used as DR_ENV padding). */
#define gp0_word_nop() enc_gp0_cmd_word(gp0_cmd_Nop)
/* ============================================================================ /* ============================================================================
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts) * GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
@@ -184,58 +185,57 @@ enum {
* (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`). * (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`).
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp1_cmd_Reset = 0x00, gp1_cmd_Reset = 0x00,
gp1_cmd_ResetCmdBuffer = 0x01, gp1_cmd_ResetCmdBuffer = 0x01,
gp1_cmd_AcknowledgeIRQ = 0x02, gp1_cmd_AcknowledgeIRQ = 0x02,
gp1_cmd_DisplayEnable = 0x03, gp1_cmd_DisplayEnable = 0x03,
gp1_cmd_DMADirection = 0x04, gp1_cmd_DMADirection = 0x04,
gp1_cmd_StartDisplayArea = 0x05, gp1_cmd_StartDisplayArea = 0x05,
gp1_cmd_HorizontalDisplayRange = 0x06, gp1_cmd_HorizontalDisplayRange = 0x06,
gp1_cmd_VerticalDisplayRange = 0x07, gp1_cmd_VerticalDisplayRange = 0x07,
gp1_cmd_DisplayMode = 0x08, gp1_cmd_DisplayMode = 0x08,
/* Note: GP1 only has commands 0x00..0x08. /* Note: GP1 only has commands 0x00..0x08.
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit) * The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
* live in the GP0 enum as * 0xE1..0xE6. * live in the GP0 enum as * 0xE1..0xE6.
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */ * DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode"). /* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
* Bit positions match the encoder shifts below; values are the * Bit positions match the encoder shifts below; values are the *payload* bits only (cmd byte is OR'd in by enc_gp1_disp_mode_word). */
* *payload* bits only (the cmd byte is OR'd in by enc_gp1_disp_mode_word). */ gp1_disp_HRes_256 = 0x0,
gp1_disp_HRes_256 = 0x0, gp1_disp_HRes_320 = 0x1,
gp1_disp_HRes_320 = 0x1, gp1_disp_HRes_512 = 0x2,
gp1_disp_HRes_512 = 0x2, gp1_disp_HRes_640 = 0x3,
gp1_disp_HRes_640 = 0x3, gp1_disp_VRes_240 = 0x0,
gp1_disp_VRes_240 = 0x0, gp1_disp_VRes_480 = 0x1,
gp1_disp_VRes_480 = 0x1, gp1_disp_Color15 = 0x0,
gp1_disp_Color15 = 0x0, gp1_disp_Color24 = 0x1,
gp1_disp_Color24 = 0x1, gp1_disp_VInterlace = 0x1,
gp1_disp_VInterlace = 0x1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */ /* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3, gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1, gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1, gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1, gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */ /* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF, gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF, gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */ /* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF, gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF, gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y /* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
* (10-bit signed — caller pre-signs and masks with the named mask) */ * (10-bit signed — caller pre-signs and masks with the named mask) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF, gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF, gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
}; };
/* ---- Layer 1.5: GP1 per-field encoders ---- */ /* ---- Layer 1.5: GP1 per-field encoders ---- */
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift) #define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift) #define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift) #define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask << gp1_disp_interlace_shift) #define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift) #define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift) #define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
@@ -255,6 +255,11 @@ enum {
#define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y)) #define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
/* ---- Layer 3: GP1 semantic word builders ---- */ /* ---- Layer 3: GP1 semantic word builders ---- */
#define gp1_word_Reset() enc_gp0_cmd_word(gp1_cmd_Reset)
#define gp1_word_ResetCmdBuffer() enc_gp0_cmd_word(gp1_cmd_ResetCmdBuffer)
#define gp1_word_AcknowledgeIRQ() enc_gp0_cmd_word(gp1_cmd_AcknowledgeIRQ)
#define gp1_word_StartDisplayArea() enc_gp0_cmd_word(gp1_cmd_StartDisplayArea)
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1)) #define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
#define gp1_word_display_disable() gp1_word_display_enable(0) #define gp1_word_display_disable() gp1_word_display_enable(0)
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0) #define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
@@ -279,31 +284,36 @@ enum {
#define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable) #define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable)
#define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1) #define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1)
#define gp1_word_DisplayOn() gp1_word_display_enable(0)
#define gp1_word_DisplayOff() gp1_word_display_enable(1)
/* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */ /* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */
enum { enum {
gp1_dma_dir_Off = 0, gp1_dma_dir_Off = 0,
gp1_dma_dir_FIFO = 1, gp1_dma_dir_FIFO = 1,
gp1_dma_dir_CPU_to_GPU = 2, gp1_dma_dir_CPU_to_GPU = 2,
gp1_dma_dir_GPUREAD_to_CPU = 3, gp1_dma_dir_GPUREAD_to_CPU = 3,
}; };
#define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3)) #define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3))
#define gp1_word_dma_to_gpu() gp1_word_dma_direction(gp1_dma_dir_CPU_to_GPU)
#define gp1_word_dma_read_cpu() gp1_word_dma_direction(gp1_dma_dir_GPUREAD_to_CPU)
/* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */ /* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */
/* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */ /* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */
enum { enum {
/* NTSC horizontal range: X1=608, X2=3168 */ /* NTSC horizontal range: X1=608, X2=3168 */
gp1_hrange_NTSC_x1 = 0x260, gp1_hrange_NTSC_x1 = 0x260,
gp1_hrange_NTSC_x2 = 0xC60, gp1_hrange_NTSC_x2 = 0xC60,
/* PAL horizontal range (same as NTSC for most CRTs) */ /* PAL horizontal range (same as NTSC for most CRTs) */
gp1_hrange_PAL_x1 = 0x260, gp1_hrange_PAL_x1 = 0x260,
gp1_hrange_PAL_x2 = 0xC60, gp1_hrange_PAL_x2 = 0xC60,
/* NTSC vertical range: Y1=24, Y2=264 */ /* NTSC vertical range: Y1=24, Y2=264 */
gp1_vrange_NTSC_y1 = 24, gp1_vrange_NTSC_y1 = 24,
gp1_vrange_NTSC_y2 = 264, gp1_vrange_NTSC_y2 = 264,
/* PAL vertical range: Y1=24, Y2=504 */ /* PAL vertical range: Y1=24, Y2=504 */
gp1_vrange_PAL_y1 = 24, gp1_vrange_PAL_y1 = 24,
gp1_vrange_PAL_y2 = 504, gp1_vrange_PAL_y2 = 504,
}; };
#define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2) #define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2)
@@ -314,14 +324,49 @@ enum {
/* ---- Draw-mode setting (TPage / draw-area allowance) ---- */ /* ---- Draw-mode setting (TPage / draw-area allowance) ---- */
/* The "drawing enabled" word is the standard post-init state. */ /* The "drawing enabled" word is the standard post-init state. */
enum { enum {
gp0_DrawMode_DrawToDispBit = 10, /* Per psx-spx, the standard 0xE1 layout has dfe at bit 10. But libpsyx's PutDrawEnv
* uses bit 19 (in the "unused" 14-23 range) for dfe in the DR_ENV code[0] — and the
* PSX hardware honors bit 19 in the DR_ENV context (not bit 10). So we need a
* separate bit definition for the DR_ENV-specific DrawMode. */
gp0_DrawMode_DrawToDispBit = 10, // standard psx-spx bit 10 (dfe)
gp0_DrawMode_DR_ENV_DrawToDispBit = 19, // libpsyx DR_ENV code[0] (dfe in DR_ENV context)
gp0_DrawMode_DR_ENV_isbgBit = 19, // libpsyx uses bit 19 for isbg too
}; };
#define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit)) #define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit))
/* ---- DrawArea pre-baked at origin (0,0) and full screen (320x240) ---- */ /* DR_ENV-specific DrawMode variants (libpsyx SetDrawEnv layout).
* The DR_ENV is a 16-word packet emitted at boot by gp_screen_init's ac_put_draw_env_demo
* atom component. Within the DR_ENV, the 0xE1 command is reused in three different bit
* configurations:
* code[0] = `gp0_word_draw_mode_drawing_allowed` (dfe=1; standard post-init state)
* code[6] = `gp0_word_dr_env_bg_color_cmd(isbg, r, g, b)` (initial-bg-color path)
* code[7] = `gp0_word_dr_env_draw_mode(isbg)` (isbg-flag path)
* Bits 0-23 of the 0xE1 word are the payload; bits 24-31 are the cmd byte (0xE1). */
#define gp0_word_dr_env_bg_color_cmd(isbg, r, g, b) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
#define gp0_word_dr_env_draw_mode(isbg) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0))
/* State-setter bare-cmd words (no immediate payload; the GPU uses the current state machine already programmed). */
#define gp0_word_set_texture_window() enc_gp0_cmd_word(gp0_cmd_SetTextureWindow)
#define gp0_word_set_draw_offset() enc_gp0_cmd_word(gp0_cmd_SetDrawOffset)
#define gp0_word_set_mask_bit() enc_gp0_cmd_word(gp0_cmd_SetMaskBit)
/* DR_ENV code[5] Mask (0xE6 cmd + isbg bit). The isbg bit is set so the GPU knows the auto-clear path is active (paired with code[6] + code[7]). */
#define gp0_word_dr_env_mask() (gp0_word_set_mask_bit() | gp0_dr_env_isbg_bit)
/* DR_ENV pre-baked constants (libpsyx PutDrawEnv layout).
* DR_ENV is a 16-word packet: tag = (length << 24) | addr, where length = 15 (15 code words follow) and addr = 0 (chain to nothing). */
enum {
PolyTag_len_bits = 8,
PolyTag_addr_bits = 24,
gp0_dr_env_tag = (15 << 24) | 0x00FFFFFF,
gp0_dr_env_isbg_bit = (1 << gp0_DrawMode_DR_ENV_isbgBit),
};
/* ---- DrawArea at origin (0,0) and full screen (320x240) ---- */
#define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0) #define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0)
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(320, 240) #define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(319, 239)
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(640, 480) #define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(639, 479)
#pragma endregion GPU Ports & Commands #pragma endregion GPU Ports & Commands
@@ -332,9 +377,9 @@ enum {
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width). * Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp1_Status_BitReady = 31, gp1_Status_BitReady = 31,
gp1_Status_BitSendingDMA = 25, gp1_Status_BitSendingDMA = 25,
gp1_Status_DMABlockSizeShift = 0, gp1_Status_DMABlockSizeShift = 0,
}; };
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1) #define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
@@ -360,10 +405,10 @@ typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
#define rgb8(r,g,b) ((RGB8){r,g,b}) #define rgb8(r,g,b) ((RGB8){r,g,b})
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */ /* ---------- PolyTag (the OT-link header; 1 word) ---------- */
enum { // enum {
PolyTag_len_bits = 8, // PolyTag_len_bits = 8,
PolyTag_addr_bits = 24, // PolyTag_addr_bits = 24,
}; // };
typedef Struct_(PolyTag) { typedef Struct_(PolyTag) {
union { union {
U4 code; U4 code;
@@ -387,95 +432,95 @@ typedef Struct_(PolyTag) {
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */ /* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
typedef Struct_(Poly_F3) { typedef Struct_(Poly_F3) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
union { union {
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; }; struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
A3_V2_S2 points; A3_V2_S2 points;
}; };
}; };
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */ /* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
typedef Struct_(Poly_F4) { typedef Struct_(Poly_F4) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
union { union {
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; }; struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
A4_V2_S2 points; A4_V2_S2 points;
}; };
}; };
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */ /* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
typedef Struct_(Poly_G3) { typedef Struct_(Poly_G3) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; V2_S2 p2;
}; };
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */ /* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
typedef Struct_(Poly_G4) { typedef Struct_(Poly_G4) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; RGB8 c3; B1 pad3; V2_S2 p2; RGB8 c3; B1 pad3;
V2_S2 p3; V2_S2 p3;
}; };
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */ /* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */ /* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
typedef Struct_(Poly_FT3) { typedef Struct_(Poly_FT3) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 p0; U1 u0; U1 v0; V2_S2 p0; U1 u0; U1 v0;
V2_S2 p1; U1 u1; U1 v1; V2_S2 p1; U1 u1; U1 v1;
V2_S2 p2; U1 u2; U1 v2; V2_S2 p2; U1 u2; U1 v2;
}; };
/* ---------- Poly_FT4 (Flat Textured Quad) ---------- */ /* ---------- Poly_FT4 (Flat Textured Quad) ---------- */
typedef Struct_(Poly_FT4) { typedef Struct_(Poly_FT4) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 p0; U1 u0; U1 v0; V2_S2 p0; U1 u0; U1 v0;
V2_S2 p1; U1 u1; U1 v1; V2_S2 p1; U1 u1; U1 v1;
V2_S2 p2; U1 u2; U1 v2; V2_S2 p2; U1 u2; U1 v2;
V2_S2 p3; U1 u3; U1 v3; V2_S2 p3; U1 u3; U1 v3;
}; };
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */ /* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
typedef Struct_(Poly_GT3) { typedef Struct_(Poly_GT3) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; V2_S2 p2;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 tp0; U1 u0; U1 v0; V2_S2 tp0; U1 u0; U1 v0;
V2_S2 tp1; U1 u1; U1 v1; V2_S2 tp1; U1 u1; U1 v1;
V2_S2 tp2; U1 u2; U1 v2; V2_S2 tp2; U1 u2; U1 v2;
}; };
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */ /* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
typedef Struct_(Poly_GT4) { typedef Struct_(Poly_GT4) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; RGB8 c3; B1 pad3; V2_S2 p2; RGB8 c3; B1 pad3;
V2_S2 p3; V2_S2 p3;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 tp0; U1 u0; U1 v0; V2_S2 tp0; U1 u0; U1 v0;
V2_S2 tp1; U1 u1; U1 v1; V2_S2 tp1; U1 u1; U1 v1;
V2_S2 tp2; U1 u2; U1 v2; V2_S2 tp2; U1 u2; U1 v2;
V2_S2 tp3; U1 u3; U1 v3; V2_S2 tp3; U1 u3; U1 v3;
}; };
/* ---------- Primitive setters (C-level) ---------- /* ---------- Primitive setters (C-level) ----------
@@ -510,26 +555,29 @@ typedef Struct_(Poly_GT4) {
* bits 12..31 = reserved (zero) * bits 12..31 = reserved (zero)
* ============================================================================ */ * ============================================================================ */
enum { enum {
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */ /* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF, gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1, gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3, gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3, gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1, gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1, gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1, gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
/* TPage color-depth payload values (NOT bit positions — these go in /* TPage color-depth payload values (NOT bit positions — these go in
* the 2-bit field at gp0_tpage_color_depth_shift). */ * the 2-bit field at gp0_tpage_color_depth_shift). */
gp0_tpage_color_4bpp = 0x0, gp0_tpage_color_4bpp = 0x0,
gp0_tpage_color_8bpp = 0x1, gp0_tpage_color_8bpp = 0x1,
gp0_tpage_color_16bpp = 0x2, gp0_tpage_color_16bpp = 0x2,
/* TPage semi-transparency mode payload values (NOT bit positions). */ /* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
gp0_tpage_semi_trans_none = 0x0, gp0_tpage_default = 10,
gp0_tpage_semi_trans_alpha = 0x1,
gp0_tpage_semi_trans_add = 0x2, /* TPage semi-transparency mode payload values (NOT bit positions). */
gp0_tpage_semi_trans_sub = 0x3, gp0_tpage_semi_trans_none = 0x0,
gp0_tpage_semi_trans_alpha = 0x1,
gp0_tpage_semi_trans_add = 0x2,
gp0_tpage_semi_trans_sub = 0x3,
}; };
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */ /* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
@@ -543,19 +591,19 @@ enum {
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */ /* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \ #define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
(enc_gp0_tpage_x(x) \ (enc_gp0_tpage_x(x) \
| enc_gp0_tpage_y(y) \ | enc_gp0_tpage_y(y) \
| enc_gp0_tpage_semi_trans(semi_trans) \ | enc_gp0_tpage_semi_trans(semi_trans) \
| enc_gp0_tpage_color_depth(color_depth) \ | enc_gp0_tpage_color_depth(color_depth) \
| enc_gp0_tpage_dither(dither) \ | enc_gp0_tpage_dither(dither) \
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \ | enc_gp0_tpage_draw_to_disp(draw_to_disp) \
| enc_gp0_tpage_tex_disable(tex_disable)) | enc_gp0_tpage_tex_disable(tex_disable))
typedef Struct_(TexturePage) { U4 raw; }; typedef Struct_(TexturePage) { U4 raw; };
/* ---- Layer 3: TPage semantic word builder ---- */ /* ---- Layer 3: TPage semantic word builder ---- */
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \ #define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable)) enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
#pragma endregion TPage #pragma endregion TPage
#pragma region CLUT #pragma region CLUT
@@ -569,12 +617,12 @@ typedef Struct_(TexturePage) { U4 raw; };
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load) * bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
* ============================================================================ */ * ============================================================================ */
enum { enum {
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */ /* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F, gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF, gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */ /* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
gp0_clut_cmd_Load4bpp = 0x20, gp0_clut_cmd_Load4bpp = 0x20,
gp0_clut_cmd_Load8bpp = 0x25, gp0_clut_cmd_Load8bpp = 0x25,
}; };
/* ---- Layer 1.5: CLUT per-field encoders ---- */ /* ---- Layer 1.5: CLUT per-field encoders ---- */
@@ -615,26 +663,26 @@ enum {
* Stoppped for now at the struct + enum level. * Stoppped for now at the struct + enum level.
* ============================================================================ */ * ============================================================================ */
enum { enum {
tim_file_id_magic = 0x10, tim_file_id_magic = 0x10,
tim_type_4bpp = 0x00, tim_type_4bpp = 0x00,
tim_type_8bpp = 0x01, tim_type_8bpp = 0x01,
tim_type_16bpp = 0x02, tim_type_16bpp = 0x02,
tim_type_32bpp = 0x03, tim_type_32bpp = 0x03,
tim_type_mixed = 0x04, tim_type_mixed = 0x04,
tim_flag_has_clut = 0x08, tim_flag_has_clut = 0x08,
}; };
typedef Struct_(TIM_Header) { typedef Struct_(TIM_Header) {
U4 file_id; /* always 0x10 = "TIM" magic */ U4 file_id; /* always 0x10 = "TIM" magic */
U4 version; /* ignored; always 0 */ U4 version; /* ignored; always 0 */
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */ U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
}; };
typedef Struct_(TIM_SectionHeader) { typedef Struct_(TIM_SectionHeader) {
U4 section_length; /* bytes in this section including this header */ U4 section_length; /* bytes in this section including this header */
U2 org_x; /* origin in VRAM */ U2 org_x; /* origin in VRAM */
U2 org_y; U2 org_y;
U2 width; /* width in pixels */ U2 width; /* width in pixels */
U2 height; /* height in pixels */ U2 height; /* height in pixels */
}; };
#pragma endregion TIM File Format #pragma endregion TIM File Format
+293
View File
@@ -0,0 +1,293 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "gte.h"
# include "gp.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
#pragma region MACs (Mips Atom Components)
/* Words: 3; Loads 3 S2 indices from the face array */
FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, {
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
})
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
FI_ Slice_MipsCode ac_gte_store_f3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
})
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, {
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
})
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
})
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
* SXY0 still holds v0.screen from the earlier RTPT.
*/
FI_ Slice_MipsCode ac_gte_store_g4_p3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
* Stage 2 of normalize consumes these directly.
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop, gte_cmdw_sqr,
gte_mv_from_data_r(r_sq_x, C2_MAC1),
gte_mv_from_data_r(r_sq_y, C2_MAC2),
gte_mv_from_data_r(r_sq_z, C2_MAC3),
})
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
* Used standalone for "scale vector by scalar".
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, {
gte_mv_to_data_r(r_recip_est, C2_IR0),
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
gte_cmdw_gpf,
gte_mv_from_data_r(r_dx, C2_MAC1),
gte_mv_from_data_r(r_dy, C2_MAC2),
gte_mv_from_data_r(r_dz, C2_MAC3),
shift_aright_var(r_dx, r_dx, r_shift),
shift_aright_var(r_dy, r_dy, r_shift),
shift_aright_var(r_dz, r_dz, r_shift),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Atom Procs
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
*
* Data is identical to the libgte original (byte-for-byte verified).
*
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
* Sampling the first value of each octave:
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
*
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
* and the load upper_halves of the table bracket the input range.
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
*
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
};
/* ─── Full normalize (all 4 stages inline) ───
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav).
*
* Parameterized by caller-provided scratch base + src/dst offsets.
* The caller passes r_src_offset and r_dst_offset as compile-time constants
* (typically derived from O_ macros in the caller's struct schema, e.g., `O_(CallerBundleScratch, fwd)`).
*
* This design lets any caller (with a scratch base + struct schema) use `normalize_v3s4_proc`
* without putting magic offsets in the C-side bundle helper — the offsets come from O_ macros at the call site.
*
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
* r_tmp : scratch (reserved for misc use)
* r_mac1_scratch : MAC1 result scratch (before sum into r_recip_est)
* r_mac2_scratch : MAC2 result scratch (clobbered to IR1 in stage 4)
* r_recip_est : |v|² sum + shift-input + sqrtbl[index] (the main chain)
* r_lzcr : LZCR value (consumed by stage 3 alignment calc)
* r_shift : final srav amount (consumed by stage 4 shift_aright_var)
* r_branch_tmp : scratch (shift count, branch target, sqrtbl base addr)
*
* Atom_labels are srav_path / aligned_done
* (NOT namespaced — they're internal to this proc;
* the metaprogram's per-atom-name enum emission handles any collision across different atoms/files that share the same labels).
*
* Pool cost: 11 GPRs (well within the 9-10 caller-trash GPR budget when r_scratch is a wave-context carrier).
*
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
* Words: ~59 (matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
*/
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
I_ void normalize_v3s4_proc(MipsAtomBuilder_R ab, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */
, U4 r_src_offset, U4 r_dst_offset /* GPR codes: PARAMETERIZED offsets (caller passes O_ macros) */
, U4 r_src_ptr, U4 r_dst_ptr, U4 r_tmp /* GPR codes: 3 scratch regs (src/dst computed + tmp) */
, U4 r_mac1_scratch, U4 r_mac2_scratch /* GPR codes: 2 more: MAC1/MAC2 scratch */
, U4 r_recip_est /* GPR code: |v|² sum + shift-input + sqrtbl[index] */
, U4 r_lzcr, U4 r_shift /* GPR codes: lzcr + final srav amount */
, U4 r_branch_tmp /* GPR code: scratch (shift count, branch target, lookup addr) */
)
MipsAtom_Proc_(normalize_v3s4, ab, {
add_si(r_src_ptr, r_scratch, r_src_offset), /* r_src_ptr = &src */
add_si(r_dst_ptr, r_scratch, r_dst_offset), /* r_dst_ptr = &dst */
nop,
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_mac2_scratch/r_recip_est/r_branch_tmp. */
load_word(r_mac2_scratch, r_src_ptr, O_(V3_S4,x)),
load_word(r_recip_est, r_src_ptr, O_(V3_S4,y)),
load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)),
nop, /* load-delay */
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
gte_mv_to_data_r(r_mac2_scratch, C2_IR1),
gte_mv_to_data_r(r_recip_est, C2_IR2),
gte_mv_to_data_r(r_branch_tmp, C2_IR3),
nop, gte_cmdw_sqr,
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
gte_mv_from_data_r(r_mac1_scratch, C2_MAC1),
gte_mv_from_data_r(r_mac2_scratch, C2_MAC2),
gte_mv_from_data_r(r_lzcr, C2_MAC3),
nop,
add_u(r_lzcr, r_lzcr, r_mac2_scratch),
add_u(r_lzcr, r_lzcr, r_mac1_scratch),
gte_mv_to_data_r(r_lzcr, C2_LZCS),
nop2,
gte_mv_from_data_r(r_shift, C2_LZCR),
nop,
/* Stage 3: compute srav amount (r_lzcr) + align |v|² to bit 24. */
and_i( r_shift, r_shift, -2),
li_s( r_lzcr, 31),
sub_s( r_lzcr, r_lzcr, r_shift),
shift_aright(r_lzcr, r_lzcr, 1),
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
add_si( r_branch_tmp, r_shift, -24),
branch_lt_zero(r_branch_tmp, atom_offset(srav_path, aligned_done)), nop,
jump_rel(atom_offset(aligned_done, srav_path)),
shift_lleft_var(r_lzcr, r_lzcr, r_branch_tmp), /* when r_branch_tmp < 0 (LZCR < 24): shift r_lzcr left by (24-LZCR) */
atom_label(srav_path)
li_s( r_branch_tmp, 24),
sub_s( r_branch_tmp, r_branch_tmp, r_shift),
shift_aright_var(r_lzcr, r_lzcr, r_branch_tmp), /* when r_branch_tmp >= 0 (LZCR >= 24): shift r_lzcr right by (LZCR-24) */
atom_label(aligned_done)
/* r_lzcr holds |v|² aligned to bit 24. */
add_si( r_lzcr, r_lzcr, -64),
shift_lleft(r_lzcr, r_lzcr, 1),
load_upper_i(r_branch_tmp, u4_hi(& gte_normalize_sqr_tbl)),
or_i_self( r_branch_tmp, u4_lo(& gte_normalize_sqr_tbl)),
add_u(r_branch_tmp, r_branch_tmp, r_lzcr),
load_half(r_lzcr, r_branch_tmp, 0), nop,
/* Stage 4: GPF + srav finalize (r_lzcr = srav_amount carried from stage 3). */
gte_mv_to_data_r(r_lzcr, C2_IR0),
gte_mv_to_data_r(r_mac2_scratch, C2_IR1),
gte_mv_to_data_r(r_recip_est, C2_IR2),
gte_mv_to_data_r(r_branch_tmp, C2_IR3),
nop2, gte_cmdw_gpf,
gte_mv_from_data_r(r_mac2_scratch, C2_MAC1),
gte_mv_from_data_r(r_recip_est, C2_MAC2),
gte_mv_from_data_r(r_branch_tmp, C2_MAC3),
shift_aright_var(r_mac2_scratch, r_mac2_scratch, r_lzcr),
shift_aright_var(r_recip_est, r_recip_est, r_lzcr),
shift_aright_var(r_branch_tmp, r_branch_tmp, r_lzcr),
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
store_word(r_mac2_scratch, r_dst_ptr, O_(V3_S4,x)),
store_word(r_recip_est, r_dst_ptr, O_(V3_S4,y)),
store_word(r_branch_tmp, r_dst_ptr, O_(V3_S4,z)),
mac_yield()
})
#pragma endregion Atom Procs
#pragma region Baked Atoms
typedef Struct_(Binds_SetGteMT3S2S4) {
MT3_S2S4* transform;
};
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
atom_bind(Binds_SetGteMT3S2S4)
, atom_reads(R_TapePtr)
){
/* Pop matrix address from tape into R_T3 ($11) */
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
mac_yield()
};
#pragma endregion Baked Atoms
+138 -141
View File
@@ -17,9 +17,8 @@
* gte_lw_v0_xy(base) (gte + lw + v0 + xy) * gte_lw_v0_xy(base) (gte + lw + v0 + xy)
* load_upper_i (load-upper + immediate, unique verb) * load_upper_i (load-upper + immediate, unique verb)
* *
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are * Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are NOT in this header.
* NOT in this header. They live in the opt-in `gte_vendor_sym.h` for * They are in the opt-in `gte_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
* users who prefer the textbook MIPS assembly mnemonics.
* ============================================================================ */ * ============================================================================ */
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
@@ -34,20 +33,16 @@
* gte.h — Geometry Transformation Engine (COP2) for the PS1 * gte.h — Geometry Transformation Engine (COP2) for the PS1
* ============================================================================ * ============================================================================
* *
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` * Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C.
* constants from C. No GCC inline-assembly string syntax in the code body. * No GCC inline-assembly string syntax in the code body.
* *
* STYLE NOTES * STYLE NOTES
* ----------- * -----------
* - Per-field encoders are named `enc_gte_<field>(value)` and each one * - Per-field encoders are named `enc_gte_<field>(value)` and each one self-masks its argument before shifting.
* self-masks its argument before shifting. Mirrors the `enc_op / enc_rs * Mirrors the `enc_op / enc_rs / enc_rt / ...` family in mips.h.
* / enc_rt / ...` family in mips.h. * - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of the per-field encoders, plus the COP2/CO base.
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of * - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined for the common cases so call sites read like assembly source.
* the per-field encoders, plus the COP2/CO base. * - All register/field values are enums (not `#define`s) so they show up in debugger symbol tables and IDE autocomplete.
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined
* for the common cases so call sites read like assembly source.
* - All register/field values are enums (not `#define`s) so they show up
* in debugger symbol tables and IDE autocomplete.
* *
* SEE ALSO * SEE ALSO
* -------- * --------
@@ -58,8 +53,7 @@
/* --- GTE Data Registers (Coprocessor 2) --- /* --- GTE Data Registers (Coprocessor 2) ---
* Preprocessor-visible integer ids for the COP2 data register file. * Preprocessor-visible integer ids for the COP2 data register file.
* Each enum value is bound to a parallel `_Code` `#define` so the * Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
* Same pattern as the GPR `_Code` set in mips.h. */ * Same pattern as the GPR `_Code` set in mips.h. */
#define C2_VXY0_Code 0 #define C2_VXY0_Code 0
#define C2_VZ0_Code 1 #define C2_VZ0_Code 1
@@ -167,6 +161,8 @@ enum {
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */ gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
gte_cmd_op = 0x0C, /* Outer Product */ gte_cmd_op = 0x0C, /* Outer Product */
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */ gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
/* --- GTE Command Bit-Field Layout --- /* --- GTE Command Bit-Field Layout ---
* A GTE command word (sent to COP2 with RS=1) is laid out as: * A GTE command word (sent to COP2 with RS=1) is laid out as:
@@ -177,25 +173,28 @@ enum {
* +------------+--+-----+------+------+------+------+---+--------+----------+ * +------------+--+-----+------+------+------+------+---+--------+----------+
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/ * \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
* *
* Shifts/masks below are the *bit positions* and *bit widths* of each * Shifts/masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
* configurable field, used by the ENC_GTE_CMD encoder.
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h. * Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
*/ */
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1, gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3, gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3, gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3, gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1, gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F, gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
gte_shift_fake_cmd = 20,
gte_width_fake_cmd = 5,
gte_mask_fake_cmd = 0x1F,
}; };
/* --- GTE Control Register Indices (for ctc2/cfc2) --- /* --- GTE Control Register Indices (for ctc2/cfc2) ---
* Preprocessor-visible integer ids for the COP2 control register file. * Preprocessor-visible integer ids for the COP2 control register file.
* Each enum value is bound to a parallel `_Code` `#define` so the * Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths). * Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23 are reserved/unused on real hardware, so there's a gap. */
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23
* are reserved/unused on real hardware, so there's a gap. */
#define gte_cr_RT11_Code 0 #define gte_cr_RT11_Code 0
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */ #define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */ #define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
@@ -223,8 +222,9 @@ enum {
#define gte_cr_RFC_Code 27 #define gte_cr_RFC_Code 27
#define gte_cr_GFC_Code 28 #define gte_cr_GFC_Code 28
#define gte_cr_BFC_Code 29 #define gte_cr_BFC_Code 29
#define gte_cr_OFX_Code 30 #define gte_cr_OFX_Code 24
#define gte_cr_OFY_Code 31 #define gte_cr_OFY_Code 25
#define gte_cr_H_Code 26
enum { enum {
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code, gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
@@ -246,21 +246,16 @@ enum { _C2_OPS_ = 0
/* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx). /* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx).
* *
* Spans the 2x2 {From, To} × {Data, Control} register classes that the * Spans the 2x2 {From, To} × {Data, Control} register classes that the GTE exposes:
* GTE exposes:
*
* bit 1 (0x02): register class — 0 = data, 1 = control * bit 1 (0x02): register class — 0 = data, 1 = control
* bit 2 (0x04): direction — 0 = read, 1 = write * bit 2 (0x04): direction — 0 = read, 1 = write
* *
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit * The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
* numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h * (which target the data register file on any coprocessor).
* (which target the data register file on any coprocessor). They are * They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
* re-aliased here so the four-way table reads like the spec mnemonics * and so the encoding is next to its only consumer (this header).
* (MFC2 / CFC2 / MTC2 / CTC2) and so the encoding lives next to its
* only consumer (this header).
* *
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) * Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
* live in gte_vendor_sym.h. */
enum { _C2_TX_SUBS_ = 0 enum { _C2_TX_SUBS_ = 0
, sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */ , sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */
, sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */ , sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */
@@ -270,11 +265,11 @@ enum { _C2_TX_SUBS_ = 0
/* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd /* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11] * Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2 * - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
* - rt: GPR source/dest * - rt: GPR source/dest
* - rd: COP2 register index (0..31): * - rd: COP2 register index (0..31):
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases) * data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */ * ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd)) #define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
@@ -314,31 +309,31 @@ enum { _C2_TX_SUBS_ = 0
* `swc2` is redundant when we're already inside the `gte_` namespace. * `swc2` is redundant when we're already inside the `gte_` namespace.
* gte_lw rt, base, off → lwc2 rt, off(base) * gte_lw rt, base, off → lwc2 rt, off(base)
* gte_sw rt, base, off → swc2 rt, off(base) * gte_sw rt, base, off → swc2 rt, off(base)
* For the typical user-facing vector-level load (xy + z as two * For the typical user-facing vector-level load (xy + z as two instructions),
* instructions), use the higher-level `gte_load_vN` macros below. */ * use the higher-level `gte_load_vN` macros below. */
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off) #define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off) #define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
/* GTE Command Format /* GTE Command Format
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO). * Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
* The lower 25 bits are the GTE-specific command payload. * Lower 25 bits are GTE-specific command payload.
* *
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` * The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
* pattern in mips.h: each one self-masks and shifts its own field, so a * Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
* caller can build up a GTE command piece by piece (handy for state-driven * (handy for state-driven MVMVA emitters that vary one field at a time).
* MVMVA emitters that vary one field at a time).
* *
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command * `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
* word in one go. It just ORs the per-field encoders together. */ * It just ORs the per-field encoders together. */
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25)) #define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
/* Per-field encoders. Each one does (value & mask) << shift on its own. */ /* Per-field encoders. Each one does (value & mask) << shift on its own. */
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf ) #define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx ) #define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v ) #define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv ) #define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm ) #define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd) #define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd ) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) (((x) & gte_mask_fake_cmd) << gte_shift_fake_cmd)
/* Composite: all six GTE fields + the COP2/CO base. */ /* Composite: all six GTE fields + the COP2/CO base. */
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \ #define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
@@ -359,12 +354,13 @@ enum { _C2_TX_SUBS_ = 0
* Decomposition (per the `enc_gte_<field>` definitions above): * Decomposition (per the `enc_gte_<field>` definitions above):
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>) * gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
* The SF/MX/V/CV/LM fields are all zero in the common cases * The SF / MX / V / CV / LM fields are all zero in the common cases
* (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp), * (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
* so the only varying bits are the `cmd` field. * so the only varying bits are the `cmd` field.
* *
* Naming follows the file's convention: `gte_cmd_*` is the raw 6-bit `cmd` field id, `gte_cmdw_*` * Naming convention:
* is the fully-encoded 32-bit instruction word ready to drop into a `.word` directive. * - `gte_cmd_*` : Raw 6-bit `cmd` field id
* - `gte_cmdw_* : 32-bit instruction word ready to drop into a `.word` directive.
* *
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
* PsyQ-compatibility note (RTPS/RTPT): * PsyQ-compatibility note (RTPS/RTPT):
@@ -375,12 +371,12 @@ enum { _C2_TX_SUBS_ = 0
* (the perspective divide happens regardless of `sf`). * (the perspective divide happens regardless of `sf`).
* *
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear), * If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops * PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided), * The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
* `nclip` ends up wrong, and the triangle is culled. * `nclip` ends up wrong, and the triangle is culled.
* *
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years. * So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source. * NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
*/ */
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer)) #define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
@@ -390,11 +386,45 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip)) #define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op )) #define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */ #define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */ #define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva)) #define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
/* SQR / GPF cosmetic-bits compat helpers.
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
* The hardware ignores these bits (per PSX-SPX line 48). */
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
/* SQR — Square Vector.
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
* bit 19 sf=0
* bit 10 lm=1
* bits 5-0 cmd=0x28=SQR
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
/* GPF — General-purpose Interpolation.
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf*12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
* bit 19 sf=0
* bit 10 lm=0
* bits 5-0 cmd=0x3D=GPF
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps #define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt #define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */ /* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20) #define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
@@ -413,20 +443,16 @@ enum { _C2_TX_SUBS_ = 0
/** /**
* @brief Loads a single SVECTOR to GTE vector register V0 * @brief Loads a single SVECTOR to GTE vector register V0
*
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0 * @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`. * (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
* *
* Uses string-style GCC inline asm with `%0` substitution because the * Uses string-style GCC inline asm with `%0` substitution because the base register `r0` is a runtime GPR chosen by the compiler.
* base register `r0` is a runtime GPR chosen by the compiler.
* It cannot be encoded into a static `.word` constant. * It cannot be encoded into a static `.word` constant.
* *
* Usage: * Usage: asm_gte_load_v0(svector_ptr);
* asm_gte_load_v0(svector_ptr);
*/ */
/* lwc2 encoding helpers parameterized on the base GPR. /* lwc2 encoding helpers parameterized on the base GPR.
*
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0 * gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0 * gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1 * gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
@@ -435,8 +461,7 @@ enum { _C2_TX_SUBS_ = 0
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2 * gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
* *
* `base` is the GPR number to bake into the .word constant's `rs` field. * `base` is the GPR number to bake into the .word constant's `rs` field.
* These are pure compile-time integers; the C compiler constant-folds * These are pure compile-time integers; the C compiler constant-folds them into .word directives. */
* them into .word directives. */
enum { enum {
GTE_Z_Offset = 4 GTE_Z_Offset = 4
@@ -450,7 +475,6 @@ enum {
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset) #define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders /* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
*
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in * Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern). * (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
* *
@@ -459,8 +483,8 @@ enum {
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12 * gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
* *
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in), * Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
* which is exactly the register the .word constants expect. A `"$12"` clobber would conflict with the register-variable binding * which is exactly the register the .word constants expect.
* ("asm specifier for variable conflicts with asm clobber list"), so we omit it. * A `"$12"` clobber would conflict with the register-variable binding ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
* The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile. * The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
* *
* WHICH REGISTER TO PICK * WHICH REGISTER TO PICK
@@ -499,10 +523,8 @@ enum {
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt. /* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
* *
* Loads all three GTE input vectors (6 words) from three separate pointers, * Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register,
* one per GTE vector register, each loaded from its own base GPR. * each loaded from its own base GPR. Caller must bind each `pN` to `bN` via a register variable.
* Caller must bind each `pN` to `bN` via a register variable.
*
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12") * register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13") * register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14") * register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
@@ -521,29 +543,20 @@ enum {
/** /**
* @brief Rotate, Translate and Perspective Triple (23 cycles) * @brief Rotate, Translate and Perspective Triple (23 cycles)
* * @details Performs rotation, translation and perspective calculation of three vertices at once.
* @details Performs rotation, translation and perspective calculation of three * The equation performed is the same as gte_rtps() only repeated three times for each vertex.
* vertices at once. The equation performed is the same as gte_rtps() only * The result of the first vertex is stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then C2_SXY2.
* repeated three times for each vertex. The result of the first vertex is
* stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then
* C2_SXY2.
* *
* Encoder-style emission (no inline-asm strings in the code body): * Encoder-style emission (no inline-asm strings in the code body):
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE * 1. Two `nop` words fill the COP2 pipeline latency — the GTE takes ~8 cycles per perspective divide,
* takes ~8 cycles per perspective divide, and the nops let any * and the nops let any preceding lwc2/swc2 retire before RTPT starts reading its inputs from V0/V1/V2.
* preceding lwc2/swc2 retire before RTPT starts reading its * 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the pre-baked encoders above) —
* inputs from V0/V1/V2. * `0x0280030` decoded as `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields zero
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the * (standard rotation, no scaling, V0 vector, translation vector, no clamp).
* pre-baked encoders above) — `0x0280030` decoded as
* `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields
* zero (standard rotation, no scaling, V0 vector, translation
* vector, no clamp).
* *
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel * Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI)
* ABI) plus the standard "memory" barrier. Does not clobber any COP2 * plus the standard "memory" barrier. Does not clobber any COP2 data/control register —
* data/control register — those have to be saved by the caller if * those have to be saved by the caller if they need to survive across the call (RTPT writes SXY0..2, SZ0..3, OTZ, MAC0..3, IR0..3, etc.).
* they need to survive across the call (RTPT writes SXY0..2, SZ0..3,
* OTZ, MAC0..3, IR0..3, etc.).
*/ */
#define gte_rtpt() \ #define gte_rtpt() \
asm volatile( \ asm volatile( \
@@ -559,32 +572,24 @@ enum {
/** /**
* @brief Normal clipping (8 cycles) * @brief Normal clipping (8 cycles)
* * @details Computes the sign of three screen coordinates (C2_SXY0-2) used for backface culling.
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for * If the value of C2_MAC0 is negative, the coordinates are inverted and thus the triangle is back facing.
* backface culling. If the value of C2_MAC0 is negative, the coordinates are
* inverted and thus the triangle is back facing.
* *
* The following equation is performed when executing this GTE command: * The following equation is performed when executing this GTE command:
*
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1 * MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
*
* Encoder-style emission (no inline-asm strings in the code body): * Encoder-style emission (no inline-asm strings in the code body):
* 1. Two `nop` words fill the COP2 pipeline latency - the GTE * 1. Two `nop` words fill the COP2 pipeline latency
* pipeline takes a few cycles per op, and the nops let any * - the GTE pipeline takes a few cycles per op, and the nops let any preceding
* preceding lwc2/swc2/RTPT retire before NCLIP starts reading * lwc2/swc2/RTPT retire before NCLIP starts reading its inputs from SXY0/SXY1/SXY2.
* its inputs from SXY0/SXY1/SXY2. * 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the pre-baked encoders above)
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the * - `0x01400006` decoded as `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields zero.
* pre-baked encoders above) - `0x01400006` decoded as * NCLIP is spec-clean in the original PsyQ source (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat` quirk),
* `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields * so `gte_cmdw_nclip` does NOT OR in any reserved bits.
* zero. NCLIP is spec-clean in the original PsyQ source
* (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat`
* quirk), so `gte_cmdw_nclip` does NOT OR in any reserved bits.
* *
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel * Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI) plus the standard "memory" barrier.
* ABI) plus the standard "memory" barrier. Does not clobber any COP2 * Does not clobber any COP2 data/control register.
* data/control register - those have to be saved by the caller if * Those have to be saved by the caller if they need to survive across the call (NCLIP writes MAC0 only;
* they need to survive across the call (NCLIP writes MAC0 only; it * it is purely a sign-of-double-product computation on SXY0..2).
* is purely a sign-of-double-product computation on SXY0..2).
*/ */
#define gte_nclip() \ #define gte_nclip() \
asm volatile( \ asm volatile( \
@@ -610,13 +615,10 @@ enum {
"cop2 0x0158002D;") "cop2 0x0158002D;")
/* asm_gte_matrix_set_rotation(r0) /* asm_gte_matrix_set_rotation(r0)
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
* *
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix * Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
* control registers (RT11..RT22, indices 0..4) via ctc2. * The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as
*
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16),
* each holding two packed 16-bit matrix elements. The first 1.5 rows
* of a standard PSX SDK MATRIX struct (where each row is laid out as
* [RT_xx, RT_xy] | [RT_xz, pad] | ...). * [RT_xx, RT_xy] | [RT_xz, pad] | ...).
* *
* Generated MIPS (mirrors the source macro): * Generated MIPS (mirrors the source macro):
@@ -631,27 +633,22 @@ enum {
* ctc2 $13, $3 ; → C2_RT21 * ctc2 $13, $3 ; → C2_RT21
* ctc2 $14, $4 ; → C2_RT22 * ctc2 $14, $4 ; → C2_RT22
* *
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a * Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
* register variable (`rgcc(R_T4)`) for the `lw $12, off(...)` * instructions to read from the right base. The `"r"(r0)` constraint alone doesn't force a specific GPR — it just lets GCC pick one.
* instructions to read from the right base. The `"r"(r0)` constraint * The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field of each lw, so the lw instructions will
* alone doesn't force a specific GPR — it just lets GCC pick one. * only do the right thing if $12 / $13 / $14 hold the matrix base at runtime.
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field
* of each lw, so the lw instructions will only do the right thing
* if $12/$13/$14 hold the matrix base at runtime.
* *
* M3_S2* m = ...; * M3_S2* m = ...;
* register M3_S2* m_in_12 rgcc(R_T4) = m; * register M3_S2* m_in_12 rgcc(R_T4) = m;
* asm_gte_matrix_set_rotation(m_in_12); * asm_gte_matrix_set_rotation(m_in_12);
* *
* We clobber $12/$13/$14 (the ones we use as scratch inside the * We clobber $12/$13/$14 (the ones we use as scratch inside the inline asm)
* inline asm) plus the system clobbers; we don't clobber `r0` because * plus the system clobbers; we don't clobber `r0` because the `rgcc` binding already says "this variable lives in $12".
* the `rgcc` binding already says "this variable lives in $12".
* *
* WARNING: Incomplete by design. The source macro only writes RT11..RT22 * WARNING: Incomplete by design. The source macro only writes RT11..RT22 (5 of 9 rotation elements);
* (5 of 9 rotation elements); RT23 and the entire RT3x row are left * RT23 and the entire RT3x row are left untouched.
* untouched. Real libpsn00b SetRotMatrix writes all 9. Use only when the * Real libpsn00b SetRotMatrix writes all 9. Use only when the GTE's remaining rotation entries are already correct,
* GTE's remaining rotation entries are already correct, or you will * or you will get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
* get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
*/ */
#define asm_gte_matrix_set_rotation(r0) \ #define asm_gte_matrix_set_rotation(r0) \
asm volatile( \ asm volatile( \
+201 -278
View File
@@ -1,100 +1,206 @@
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
# pragma once # pragma once
# include "gen/macs.h"
# include "gen/offsets.h"
# include "dsl.h" # include "dsl.h"
# include "gcc_asm.h" # include "gcc_asm.h"
# include "mips.h" # include "mips.h"
# include "gte.h" # include "gte.h"
# include "memory.h" # include "memory.h"
# include "atom_dsl.h" # include "dsl.atom.h"
# include "gen/duffle.macs.h"
# include "gen/duffle.offsets.h"
#endif #endif
typedef U4 const MipsCode; #pragma region Tape Drive
typedef Slice_(MipsCode); /* -----------------------------------------------------------------------------------------------------------
typedef Slice_MipsCode MipsAtom; * TAPE DRIVE ABI
* -----------------------------------------------------------------------------------------------------------
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
* as such the information below may not* be entirely realized or finalized conceptually.
* -----------------------------------------------------------------------------------------------------------
* This ABI and its associated legos were directly inspired by researching the work of
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
*
* This behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
* to author and compose programs with. From here various conventions can be further applied.
* To make things easier to understand it may be better to focus on what this ABI does not have.
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
* Branching nearly is always downstream. Atuomatic stack usage is non-existent.
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
*
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
* or, basically anything from the 5th generation consoles and onward.
* The ABI directly reflects how all computational hardware must be architected in order to execute
* digital logic effectively on current era tech.
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
* and core atoms to take those newer hardware features into account. For example, you can easily expand
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
* automatic register allocation means the user cannott ignore excessive argument shuffle across workload or
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
*
* Learning data-oreinted code becomes a natural progression. Your not fighting a stack-based procedural
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
* it becomes a game of tetris.
*
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
* That being like a color forth, or maybe something more familar like an immediate mode library
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
* -----------------------------------------------------------------------------------------------------------
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
* -----------------------------------------------------------------------------------------------------------
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
* (just copying ram to filesystem), I can author a color forth to mess around with.
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
* */
/* Register Allocation Info */
enum {
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
/* Stringification codes for the GCC inline assembler clobber lists. */
#define R_AtomJmp_Code R_T8_Code
#define R_TapePtr_Code R_T9_Code
// R_InCursor = R_T4,
// #define R_InCursor_Code R_T4_Code
// Reserved Registers (Callee-saved):
// - R_T9: Holds the Tape Ptr which we need to increment
// If we hit a wall with register allocations we can clobber V0 & V1 (return values), defering as opt-in by user.
// - R_RA: Not sure??
// Needed by ac_yield but can be used as atom scratch:
// - R_T8: Will be used as the atom jump register.
// All allocatable registers for mips atoms:
R_TScratchVolatile = R_AT, // This one is reserved for psuedo instructions, but you can technically use it.
R_TScratch0 = R_T0,
R_TScratch1 = R_T1,
R_TScratch2 = R_T2,
R_TScratch3 = R_T3,
R_TScratch4 = R_T4,
R_TScratch5 = R_T5,
R_TScratch6 = R_T6,
R_TScratch7 = R_T7,
R_TScratch8 = R_T8,
R_TScratch10 = R_V0, // Tend to be used with gte DMAs
R_TScratch11 = R_V1, // Tend to be used with gte DMAs
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
// A 0-2
// S 0-7
};
typedef U4 const MipsCode; // Underlying type to mips asm words.
typedef Slice_(MipsCode);
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
#define MipsAtom_(sym) MipsCode sym [] align_(4) = #define MipsAtom_(sym) MipsCode sym [] align_(4) =
// Used for atoms with value-args
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
// expands to:
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; }
#define MipsAtom_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); }
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names). // Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
// MipsAtomComp_(ac_X) { body } // MipsAtomComp_(ac_X) { body }
// expands to: // expands to:
// MipsCode ac_X[] align_(4) = { body }; // MipsCode ac_X[] align_(4) = { body };
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) = #define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
// Used for components with value-args (e.g., ac_format_f3_color). // Used for components with value-args (mandatory `ab` (atom-builder) arg).
// FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body }) // FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ac_X, ab, { body })
// expands to: // expands to:
// FI_ MipsAtom ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); } // FI_ void ac_X(MipsAtomBuilder_R ab, args) {
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); } // MipsCode ac_X[] align_(4) = { body };
// atombuilder_unroll(ab, slice_from_array(MipsCode, ac_X));
// }
// The body must NOT include mac_yield() (the parent atom yields).
// Inline-only callers (the generated `mac_<name>` aliases) skip this arg via metaprogram filtering;
// escape callers (ac_<name> invoked as a function) pass a long-lived builder.
#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(ab, slice_from_array(MipsCode, sym)); }
// Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h) are included manually by the unity build. /* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
Files containing only atoms and atom components.
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
The macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
The constant is in `.rodata` and unreferenced; the linker may eliminate it.
The two-level concat + `__LINE__` suffix makes the identifier unique per call site
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
/* Register aliases */ typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
enum {
R_AtomJmp = R_T9 atom_reg, /* debug-visible; tape yield handshake scratch */
R_TapePtr = R_T8 atom_reg, /* The Instruction Stream Pointer */
R_InCursor = R_T4,
R_PrimCursor = R_T7 atom_reg atom_type(U4 *), /* VRAM output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2 *), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
R_VertBase = R_T5 atom_reg atom_type(V3_S2 *), /* Base address of the vertex array */
R_OtBase = R_T6 atom_reg atom_type(U4 *), /* Base address of the Ordering Table */
/* Stringification codes for the GCC inline assembler clobber lists. */
#define R_TapePtr_Code R_T8_Code
#define R_InCursor_Code R_T4_Code
#define R_PrimCursor_Code R_T7_Code
#define R_FaceCursor_Code R_T4_Code
#define R_VertBase_Code R_T5_Code
#define R_OtBase_Code R_T6_Code
};
#pragma region Tape Drive
/* ---------------------------------------------------------------------------
* TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below)
* ---------------------------------------------------------------------------*/
/* The 'Exit' Atom */ /* The 'Exit' Atom */
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }; atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
/* Generalized Tape Engine Runner */ // TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
NI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
/* Tape Runner (Default) */
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
asm_words( asm_words(
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */ load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */ , add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
, load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */ , call_reg( R_AtomJmp) /* jalr $t9 */
, add_ui_self(R_TapePtr, S_(MipsCode)) /* Advance tape */ , nop /* Branch delay slot */
, call_reg( R_AtomJmp) /* jalr $t9 */
, nop /* Branch delay slot */
, load_word( R_RA, R_SP, 0) /* Restore $ra from stack */
, add_ui_self(R_SP, MipsStackAlignment) /* Deallocate stack space */
) )
asm_rpins, r_use(tp) asm_rpins, r_use(tape_ptr)
asm_clobber: asm_clobber:
rlit(R_AT) rlit(R_AT),
, rlit(R_V0), rlit(R_V1) rlit(R_V0), rlit(R_V1), // We clobber these for GTE ACs (that don't expose register selection, might expose them in the future...)
, rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3) rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
/* Tell GCC the tape engine owns and destroys the workspace registers */ rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
, rlit(R_PrimCursor), rlit(R_FaceCursor), rlit(R_VertBase), rlit(R_OtBase) clb_mem_drain
, rlit(R_T9)
, clb_mem_drain
); } ); }
/* Tape Runner (Static and Arg Clobbers) */
FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
asm_words(
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
, call_reg( R_AtomJmp) /* jalr $t9 */
, nop /* Branch delay slot */
)
asm_rpins, r_use(tape_ptr)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1), rlit(R_A0), rlit(R_A1), rlit(R_A2),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
rlit(R_S0), rlit(R_S1), rlit(R_S2), rlit(R_S3), rlit(R_S4),
rlit(R_S5), rlit(R_S6), rlit(R_S7),
clb_mem_drain
); }
// Procedural authoring of tapes:
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; }; typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; } FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; } FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; } FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
#define tb_emit_(tb, atom) tb_emit(tb, atom) FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; } FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
#define tb_emit_(atom) tb_emit(& tb, atom)
#define tb_data_(field, data) tb_data(& tb, u4_(data))
FI_ Slice_MipsCode tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Slice_MipsCode){ C_(U4*,tb->ptr), tb->used }; } FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), S_slice(atoms)); tb->used += atoms.len; }
FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Slice_MipsCode){ C_(U4*,tb.ptr), tb.used }; }
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
#pragma endregion Tape Drive #pragma endregion Tape Drive
#pragma region Macro Mips Atom Components #pragma region Macro Mips Atom Components
@@ -103,119 +209,30 @@ FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Sl
* These do NOT yield. They are expanded inline inside Tape Atoms. * These do NOT yield. They are expanded inline inside Tape Atoms.
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
// The 'Yield' sequence for Tape Atoms (mac_yield). // The 'Yield' sequence for Tape Atoms (mac_yield).
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
// - mac_yield_load() + mac_yield_tail():
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
atom_dbg_skip MipsAtomComp_(ac_yield) { atom_dbg_skip MipsAtomComp_(ac_yield) {
load_word(R_AtomJmp, R_TapePtr, 0), load_word(R_AtomJmp, R_TapePtr, 0),
add_ui_self( R_TapePtr, S_(MipsCode)), add_ui_self( R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop, jump_reg( R_AtomJmp), nop,
}; };
/* Words: 3; Loads 3 S2 indices from the face array */ atom_dbg_skip MipsAtomComp_(ac_yield_load) {
atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) { load_word(R_AtomJmp, R_TapePtr, 0),
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
}; };
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
atom_dbg_skip MipsAtomComp_(ac_gte_load_tri_verts) { add_ui_self(R_TapePtr, S_(MipsCode)),
shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0), jump_reg( R_AtomJmp), nop,
shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
}; };
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
MipsAtomComp_(ac_insert_ot_tag_f3) {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, R_PrimCursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
};
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
MipsAtomComp_(ac_insert_ot_tag_g4) {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, R_PrimCursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
};
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
* byte offset. Internal helper used by the *_format_*_color macros. */
FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, R_PrimCursor, (off)),
})
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
};
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
FI_ MipsAtom ac_format_g4_color(
U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1,
U1 r2, U1 g2, U1 b2,
U1 r3, U1 g3, U1 b3)
MipsAtomComp_Proc_(ac_format_g4_color, {
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1),
mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2),
mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3),
})
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
* three registers aligned with v0/v1/v2 you must store before RTPS).
* The macro name declares the pipeline position; check #6 (GTE state-
* machine validation) verifies the call site matches the declaration. */
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
};
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
* prevents).
*/
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
#pragma endregion Macro Atom Components #pragma endregion Macro Atom Components
#pragma region Mips Atom Builder #pragma region Mips Atom Builder
// This allows for runtime procedural authoring of mips atoms. // This helps with runtime procedural authoring of mips atoms.
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; }; typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
@@ -224,133 +241,39 @@ typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 u
// Whatever the builder is writting to should most likely coresspond // Whatever the builder is writting to should most likely coresspond
// to something that can fit within instruction cache? // to something that can fit within instruction cache?
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) { FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode code) {
assert(ab->capacity - ab->used - code->len); /* code.len is in ELEMENTS (per slice_from_array convention); ab->used is also in elements
mem_copy(ab->start, u4_(code->ptr), code->len); * (the init uses `ab->used * sizeof(U4)` for byte offset arithmetic — sizeof(U4)==4==sizeof(MipsCode)).
mem_bump(ab->start, ab->capacity, & ab->used, code->len); * mem_copy needs BYTES, so we use S_slice(code) for the length. */
assert(ab->capacity - ab->used - code.len);
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot (arena accumulation) */
mem_copy(u4_(dest), u4_(code.ptr), S_slice(code));
mem_bump(ab->start, ab->capacity, & ab->used, code.len);
} }
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac)) #define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
// When done authoring, utilize this to cap-off the atom // When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
FI_ void atombuilder_end(MipsAtomBuilder_R ab) { FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield)); /* ac_yield is a MipsCode[] of 4 elements; S_(ac_yield)=bytes, array_len(ac_yield)=elements.
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield)); * ab->used is in elements, so mem_bump needs element count. */
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot */
mem_copy(u4_(dest), u4_(ac_yield), S_(ac_yield));
mem_bump(ab->start, ab->capacity, & ab->used, array_len(ac_yield));
} }
#define mipsatom_from_builder(ab) (MipsAtom){ab.start, ab.used} #define mipsatom_from_builder(ab) C_(MipsAtom*, (ab).start)
// tb_emit_builder(tb, ab) — emit the builder's atom into the tape and advance tb->used.
// Thin wrapper around tb_emit(tb, mipsatom_from_builder(ab[0])).
// Equivalent to tb_emit(tb, code_<name>) for runtime-built atoms.
FI_ void tb_emit_builder(TapeBuilder_R tb, MipsAtomBuilder_R ab) { tb_emit(tb, mipsatom_from_builder(ab[0])); }
#pragma endregion Mips Atom Builder #pragma endregion Mips Atom Builder
#pragma region Mips Atom Procs
#pragma endregion Mips Atom Procs
#pragma region Baked Mips Atoms #pragma region Baked Mips Atoms
// These atoms are resolved at compile time and are (usually) statically linked readonly data. // These atoms are resolved at compile time and are (usually) statically linked readonly data.
enum {
bios_flushcache = 0x44,
bios_table_addr = 0xA0,
};
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
* 2. $a0 = bios_flushcache (arg0)
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
* 4. jalr $t0, $ra ; call BIOS(flushcache)
* nop ; branch delay slot
* 5. lw $ra, 4($sp); jr $ra ; restore & return
* 6. sp += 8
*/
internal MipsAtom_(mips_flush_icache) {
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
jump_reg(rret_addr), // jr $ra
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
mac_yield(),
};
typedef Struct_(Binds_SetGteWorld) {
M3_S2* transform;
};
internal MipsAtom_(set_gte_world) atom_info(
atom_bind(Binds_SetGteWorld)
, atom_reads(R_TapePtr)
){
/* Pop matrix address from tape into R_T3 ($11) */
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
mac_yield()
};
/* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() };
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0),
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
store_word( R_AT, R_T7, 4),
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
add_ui( R_T1, R_0, 10),
shift_lleft_self(R_T1, S_(U4)/2),
add_u_self( R_T1, R_T6),
load_word( R_AT, R_T1, 0),
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
store_word( R_AT, R_T7, 0),
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
or_u_self( R_AT, R_V0),
store_word( R_AT, R_T1, 0),
add_ui(R_T7, R_T7, 20),
mac_yield()
};
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) {
/* Load 3 indices */
load_half_u(R_T0, R_T4, 0),
load_half_u(R_T1, R_T4, 2),
load_half_u(R_T2, R_T4, 4),
/* Load Vertices into GTE */
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
/* Run Math */
nop2, gte_cmdw_rtpt,
nop2, gte_cmdw_nclip,
nop2,
/* Advance Face Cursor and Yield */
add_ui(R_T4, R_T4, 8),
mac_yield()
};
#pragma endregion Baked Mips Atoms #pragma endregion Baked Mips Atoms
+47
View File
@@ -0,0 +1,47 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "math.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#pragma region MACs (Mips Atom Component)
FI_ Slice_MipsCode ac_load_v2s2(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, {
load_half( rs_x, r_base, O_(V3_S2,x)),
load_half( rs_y, r_base, O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_v3s4(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, {
load_word( rs_x, r_base, O_(V3_S4,x)),
load_word( rs_y, r_base, O_(V3_S4,y)),
load_word( rs_z, r_base, O_(V3_S4,z)),
})
FI_ Slice_MipsCode ac_store_v3s4(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, {
store_word(rt_x, base, offset + O_(V3_S4,x)),
store_word(rt_y, base, offset + O_(V3_S4,y)),
store_word(rt_z, base, offset + O_(V3_S4,z)),
})
FI_ Slice_MipsCode ac_sub_v3s4(MipsAtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, {
sub_s(rds_x, rds_x, rt_x),
sub_s(rds_y, rds_y, rt_y),
sub_s(rds_z, rds_z, rt_z),
})
FI_ Slice_MipsCode ac_store_rects2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
store_half(rt_height, base, offset + O_(Rect_S2,height)),
})
#pragma endregion MACs (Mips Atom Component)
+59 -11
View File
@@ -7,10 +7,23 @@
#define max(A, B) (((A) > (B)) ? (A) : (B)) #define max(A, B) (((A) > (B)) ? (A) : (B))
#define clamp_bot(X, B) max(X, B) #define clamp_bot(X, B) max(X, B)
/* Convention
<Type> ## <Width> _ <Component Type> ## <Component Width>
For types with compound data (Ex: Rotation Matrix & Translation):
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
A: Array
V: Vector
R: Range
M: Matrix
T: Translation
*/
enum { enum {
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset. v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
}; };
typedef Array_(U1, 2);
typedef Array_(U4, 2); typedef Array_(U4, 2);
typedef Array_(S2, 2); typedef Array_(S2, 2);
typedef Array_(S2, 3); typedef Array_(S2, 3);
@@ -22,24 +35,41 @@ typedef S2 A3x3_S2[3][3];
typedef Struct_(Extent2_S2) { S2 width; S2 height; }; typedef Struct_(Extent2_S2) { S2 width; S2 height; };
typedef Struct_(Extent2_S4) { S4 width; S4 height; }; typedef Struct_(Extent2_S4) { S4 width; S4 height; };
typedef Struct_(V2_U1) { U1 x; U1 y; };
typedef Struct_(V2_S2) { S2 x; S2 y; }; typedef Struct_(V2_S2) { S2 x; S2 y; };
typedef Struct_(V2_S4) { S4 x; S4 y; }; typedef Struct_(V2_S4) { S4 x; S4 y; };
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; }; typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; }; typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; typedef V3_S4 P3_S4;
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; }; typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; }; typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; }; typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
/* RGA(Lengyel) reserved names (deferred):
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
typedef Array_(V2_U1, 2);
typedef Array_(V2_S2, 2);
typedef Array_(V2_S2, 3); typedef Array_(V2_S2, 3);
typedef Array_(V2_S2, 4); typedef Array_(V2_S2, 4);
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
#define v2s2(x,y) (V2_S2){x,y} #define v2s2(x,y) (V2_S2){x,y}
#define v3s2(x,y,z) (V3_S2){x,y,z,0} #define v3s2(x,y,z) (V3_S2){x,y,z,0}
#define v3s4(x,y,z) (V3_S4){x,y,z,0} #define v3s4(x,y,z) (V3_S4){x,y,z,0}
@@ -58,10 +88,28 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[2] += b[2] >> 1; (out_a[0])[2] += b[2] >> 1;
} }
FI_ void add_v3s4(V3_S4_R out_a, V3_S4 b) { FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
add_a3s4(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); (out_a[0])[0] -= b[0];
(out_a[0])[1] -= b[1];
(out_a[0])[2] -= b[2];
} }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); (out_a[0])[0] -= b[0] >> 1;
(out_a[0])[1] -= b[1] >> 1;
(out_a[0])[2] -= b[2] >> 1;
} }
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] *= b[0];
(out_a[0])[1] *= b[1];
(out_a[0])[2] *= b[2];
}
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
+12 -12
View File
@@ -58,31 +58,31 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; }; typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 } #define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; } FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; } #define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
typedef Slice_(B1); typedef Slice_(B1);
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0) #define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
#define slice_end(slice) ((slice).ptr + (slice).len) #define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
#define S_slice(s) ((s).len * S_((s).ptr[0])) #define S_slice(s) ((s).len * S_((s).ptr[0]))
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len)) #define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a)) #define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s)) #define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter) #define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) } #define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) } #define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) / S_(type) } /* .len in elements (matches S_slice/slice_arg_from_array convention) */
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); } FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), S_slice(s)); }
#define slice_zero(s) slice_zero_(slice_to_ut(s)) #define slice_zero(s) slice_zero_(slice_to_ut(s))
FI_ void slice_copy_(Slice dest, Slice src) { FI_ void slice_copy_(Slice dest, Slice src) {
assert(dest.len >= src.len); assert(S_slice(dest) >= S_slice(src));
slice_assert(dest); slice_assert(dest);
slice_assert(src); slice_assert(src);
mem_copy(dest.ptr, src.ptr, src.len); mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
} }
#define slice_copy(dest, src) do { \ #define slice_copy(dest, src) do { \
static_assert(T_same(dest, src)); \ static_assert(T_same(dest, src)); \
@@ -98,8 +98,8 @@ typedef Slice_(U4);
typedef Opt_(farena) { U4 alignment, type_width; }; typedef Opt_(farena) { U4 alignment, type_width; };
typedef Struct_(FArena) { U4 start, capacity, used; }; typedef Struct_(FArena) { U4 start, capacity, used; };
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr); FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
arena->start = mem.ptr; arena->start = u4_(mem.ptr);
arena->capacity = mem.len; arena->capacity = S_slice(mem); /* FArena.used is in BYTES; capacity must be bytes too */
arena->used = 0; arena->used = 0;
} }
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; } FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
@@ -109,7 +109,7 @@ I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT); U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
U4 ptr = arena->start + arena->used; U4 ptr = arena->start + arena->used;
mem_bump(arena->start, arena->capacity, & arena->used, to_commit); mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
return (Slice){ ptr, to_commit }; return (Slice){ (B1*)ptr, to_commit };
} }
FI_ void farena_reset (FArena_R arena) { arena->used = 0; } FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
FI_ void farena_rewind(FArena_R arena, U4 save_point) { FI_ void farena_rewind(FArena_R arena, U4 save_point) {
+34
View File
@@ -0,0 +1,34 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "bios.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
#pragma region Baked Atoms
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
* 2. $a0 = bios_flushcache (arg0)
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
* 4. jalr $t0, $ra ; call BIOS(flushcache)
* nop ; branch delay slot
* 5. lw $ra, 4($sp); jr $ra ; restore & return
* 6. sp += 8
*/
internal MipsAtom_(mips_flush_icache) {
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
jump_reg(rret_addr), // jr $ra
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
mac_yield(),
};
#pragma endregion Baked Atoms
+34 -18
View File
@@ -336,10 +336,10 @@ enum { _BitOffsets = 0
/* Logic Opcodes */ /* Logic Opcodes */
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and) #define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or) #define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor) #define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor) #define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
#define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or) #define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or)
@@ -348,6 +348,12 @@ enum { _BitOffsets = 0
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl) #define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra) #define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
/* Shift Variable — register-shift forms.
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll) #define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt) #define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
@@ -362,10 +368,26 @@ enum { _BitOffsets = 0
/* call_reg rs — jump-and-link to register-held address; link in $ra. */ /* call_reg rs — jump-and-link to register-held address; link in $ra. */
#define call_reg(rs) jump_link((rs), R_RA) #define call_reg(rs) jump_link((rs), R_RA)
/* j target — absolute jump within the current 256MB region. */ /* j target — absolute jump within the current 256MB region.
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
*/
#define jump(off) enc_i(op_j, R_0, R_0, (off)) #define jump(off) enc_i(op_j, R_0, R_0, (off))
/* call_addr off — jump-and-link to immediate address. */ /* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
#define jump_rel(off) branch_equal(R_0, R_0, (off))
/* call_addr off — jump-and-link to immediate address.
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
*/
#define call_addr(off) enc_i(op_jal, R_0, R_0, (off)) #define call_addr(off) enc_i(op_jal, R_0, R_0, (off))
/* --- Store family (mirrors the load family) --- */ /* --- Store family (mirrors the load family) --- */
@@ -379,13 +401,7 @@ enum { _BitOffsets = 0
* sub_s / sub_u → sub / subu * sub_s / sub_u → sub / subu
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO) * mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
* div_s / div_u → div / divu (LO = quot, HI = rem) * div_s / div_u → div / divu (LO = quot, HI = rem)
* */
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
#undef add_s
#undef sub_s
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add) #define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu) #define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub) #define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
@@ -440,22 +456,22 @@ enum { _BitOffsets = 0
#define nop shift_lleft(rdiscard, rdiscard, 0) #define nop shift_lleft(rdiscard, rdiscard, 0)
#define nop2 nop, nop #define nop2 nop, nop
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm)) #define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm)) #define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi). /* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
*
* Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when: * Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when:
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR * - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
* - `imm` is not a compile-time constant and you want predictable * - `imm` is not a compile-time constant and you want predictable 2-word emission without the `__builtin_constant_p` branches.
* 2-word emission without the `__builtin_constant_p` branches.
* *
* The lo16 strategy is still chosen at expansion time on the lo half: * The lo16 strategy is still chosen at expansion time on the lo half:
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0) * lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0)
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern) * lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern)
* *
* For situations where you need to bypass even this choice * For situations where you need to bypass even this choice (e.g. to force a specific encoding for a known discontiguous high/low pair),
* (e.g. to force a specific encoding for a known discontiguous high/low pair),
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below. * see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
* Statement-level (not expression-level): emits its own `asm volatile(...)`. * Statement-level (not expression-level): emits its own `asm volatile(...)`.
*/ */
+195
View File
@@ -0,0 +1,195 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "mips.h"
# include "dsl.atom.h"
# include "lottes_tape.h"
# include "pad.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_pad_set_centered_axes(MipsAtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, {
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF),
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF),
store_word( r_scratch, r_state, O_(PadState,axes)),
})
FI_ Slice_MipsCode ac_pad_set_id_byte(MipsAtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, {
add_ui( r_id, R_0, id_value),
store_byte(r_id, r_state, O_(PadState,id)),
})
FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, {
add_ui( r_tmp, R_0, pad_status),
store_word(r_tmp, r_state, O_(PadState,status)),
})
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(MipsAtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, {
nor_u( r_buttons, r_buttons, R_0),
store_half( r_buttons, r_pad_state, O_(PadState, buttons)),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms
/* ----- pad_bios_snapshot -----
* Per-frame snapshot of one BIOS pad buffer into PadState.
* Decoder (branch ladder on raw[0] status + raw[1] id):
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
* 6. else -> Unsupported (buttons=0, axes=0x80)
*
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
*
* Register use (atom-local; no wave-context touched):
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
* R_T1 = state base (kept throughout; all stores go through R_T1)
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
* R_T4 = scratch (shifts, compares, immediate loads, store values)
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
*/
enum {
R_PadRaw = R_T0 atom_reg atom_type(U1),
R_PadState = R_T1 atom_reg atom_type(PadState*),
R_RawStatus = R_T2 atom_reg,
R_RawId = R_T3 atom_reg,
};
typedef Struct_(Binds_PadBiosSnapshot) {
PadBiosRaw* raw;
PadState* state;
};
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
) {
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
/* === Read raw[0] (status) + raw[1] (id) */
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
atom_label(disconnected) /* === Disconnected body. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
store_half( R_0, R_PadState, O_(PadState,buttons)),
mac_pad_set_centered_axes(R_PadState, R_T4),
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
jump_rel(atom_offset(disconnected, snap_end)),
/* BD-slot: load next atom's entry point (replaces the nop).
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
mac_yield_load(),
atom_label(skip_disconnected)
/* === Case 2: Pending (status == 0 && id == 0)
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
store_half( R_0, R_PadState, O_(PadState,buttons)),
mac_pad_set_centered_axes(R_PadState, R_T4),
store_byte(R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(pending, snap_end)),
mac_yield_load(),
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo), /* fills the buttons-load's delay slot (doesn't read R_T4) */
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
jump_rel(atom_offset(id_dispatch, snap_end)),
mac_yield_load(),
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
atom_label(analog_stick) /* === AnalogStick body
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
store_half( R_T5, R_PadState, O_(PadState, left)),
store_half( R_T4, R_PadState, O_(PadState, right)),
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
jump_rel(atom_offset(analog_stick, snap_end)),
mac_yield_load(),
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
atom_label(analog_pad) /* === AnalogPad body
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
store_half( R_T5, R_PadState, O_(PadState, left)),
store_half( R_T4, R_PadState, O_(PadState, right)),
store_byte( R_RawId, R_PadState, O_(PadState, id)),
jump_rel(atom_offset(analog_pad, snap_end)),
mac_yield_load(),
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
add_ui( R_T4, R_0, PadStatus_Unsupported),
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
mac_pad_set_centered_axes(R_PadState, R_T4),
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
/* Fall through to snap_end. */
atom_label(no_jump_fallthrough)
mac_yield_load(),
atom_label(snap_end)
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
mac_yield_tail(),
};
#pragma endregion Baked Atoms
+78
View File
@@ -0,0 +1,78 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "dsl.h"
# include "gcc_asm.h"
# include "mips.h"
# include "bios.h"
# include "pad.h"
#endif
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
*
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
* The C-level writes after the call re-load the pointers from their callee-saved homes.
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
// Use enums.
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
* $a0 = raw0 (rgcc-bound; survives the sequence below)
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, bios_pad_buffer_size), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, bios_pad_buffer_size), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, bios_init_pad_2), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
u1_v(raw0)[0] = 0xFF;
u1_v(raw1)[0] = 0xFF;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( rtmp_1, rdiscard, bios_start_pad_2), /* $t1 = 0x13 */
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
}
+115
View File
@@ -0,0 +1,115 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "dsl.h"
# include "math.h"
#endif
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
* Wire is active-low (0 = pressed).
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
* active-low-to-active-high inversion is applied bit-by-bit. */
typedef Enum_(U2, PadBtns) {
Bit_(Pad_Select, 0),
Bit_(Pad_L3, 1),
Bit_(Pad_R3, 2),
Bit_(Pad_Start, 3),
Bit_(Pad_Up, 4),
Bit_(Pad_Right, 5),
Bit_(Pad_Down, 6),
Bit_(Pad_Left, 7),
Bit_(Pad_L2, 8),
Bit_(Pad_R2, 9),
Bit_(Pad_L1, 10),
Bit_(Pad_R1, 11),
Bit_(Pad_Triangle, 12),
Bit_(Pad_Circle, 13),
Bit_(Pad_Cross, 14),
Bit_(Pad_Square, 15),
};
enum {
PadId_Offset = 4,
Pad0 = 0 << PadId_Offset,
Pad1 = 1 << PadId_Offset,
};
/* =============================================================================
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
* ============================================================================= */
enum {
PAD_BIOS_RAW_SIZE = 0x22,
};
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
typedef Struct_(PadBiosRaw) {
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
V2_U1 right; /* offset 4-5 (right stick x, y) */
V2_U1 left; /* offset 6-7 (left stick x, y) */
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
};
typedef Enum_(U4, PadStatus) {
PadStatus_Disconnected,
PadStatus_Digital,
PadStatus_AnalogStick,
PadStatus_AnalogPad,
PadStatus_Unsupported,
PadStatus_Pending,
PadStatus_Invalid,
};
/* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
* PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
* when the controller id does not match any known controller type.
* PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
* four byte axes at PadState.left_x through PadState.right_y. */
typedef Enum_(U1, PadRawStatus) {
PadRawStatus_Ok = 0x00,
PadRawStatus_Timeout = 0xFF,
};
typedef Enum_(U1, PadRawId) {
PadRawId_Digital = 0x41,
PadRawId_AnalogStick = 0x53,
PadRawId_AnalogPadMask = 0xF0,
PadRawId_AnalogPadValue = 0x70,
};
typedef Enum_(U1, PadUnknownId) {
PadUnknownId_Sentinel = 0xFF,
};
typedef Enum_(U4, PadAxisCentered) {
PadAxis_Centered_Hi = 0x8080,
PadAxis_Centered_Lo = 0x8080,
PadAxis_Centered_Word = 0x80808080U,
};
typedef Enum_(U1, PadDeadZone) {
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
};
typedef Struct_(PadAxes) {
V2_U1 left; /* offset 8-9 */
V2_U1 right; /* offset 10-11 */
};
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
typedef Struct_(PadState) {
PadStatus status; /* offset 0, (U4) */
PadBtns buttons; /* offset 4, */
U1 id; /* offset 6, */
byte_pad(1); /* offset 7, explicit pad to align the axes block */
union {
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
struct {
V2_U1 left; /* offset 8-9 */
V2_U1 right; /* offset 10-11 */
};
};
};
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
+7
View File
@@ -0,0 +1,7 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "psyq.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(pysq_atom_c);
+121
View File
@@ -0,0 +1,121 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "dsl.h"
# include "math.h"
# include "gp.h"
#endif
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
typedef Struct_(DrawEnv) {
Rect_S2 clip_area;
V2_S2 drawing_offset[2];
Rect_S2 texture_window;
S2 texture_page;
B1 flag_dither;
B1 flag_draw_on_display;
B1 enable_auto_clear;
RGB8 initial_bg_color;
DrawEnv_Packed dr_env; // reserved
};
typedef Struct_(DisplayEnv) {
Rect_S2 display_area;
Rect_S2 screen;
B1 vinterlace;
B1 color24;
B1 pad0;
B1 pad1;
};
typedef Array_(DrawEnv, 2);
typedef Array_(DisplayEnv, 2);
typedef Struct_(DoubleBuffer) {
A2_DrawEnv draw;
A2_DisplayEnv display;
};
DisplayEnv* displayenv_init(DisplayEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDispEnv");
DrawEnv* drawenv_init (DrawEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDrawEnv");
DisplayEnv* displayenv_put(DisplayEnv* env) asm("PutDispEnv");
DrawEnv* drawenv_put (DrawEnv* env) asm("PutDrawEnv");
U4 geom_init(void) asm("InitGeom");
void geom_set_offset(U4 x, U4 y) asm("SetGeomOffset");
void geom_set_screen(U4 h) asm("SetGeomScreen");
U4* orderingtbl_clear_reverse(U4* ot, U4 len) asm("ClearOTagR");
U4 reset_graph(U4 mode) asm("ResetGraph");
void set_display_enabled(U4 mask) asm("SetDispMask");
U4 draw_sync(U4 mode) asm("DrawSync");
U4 vsync(U4 mode) asm("VSync");
void draw_orderingtbl(U4* buf) asm("DrawOTag");
typedef Struct_(Tile) {
U4 tag;
RGB8 color;
B1 code;
Rect_S2 rect;
};
/*
Linear Algebra
*/
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
// Rotation, Translation, Perspective
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) asm("RotTransPers");
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip3");
FI_ S4 rtp_avg_nclip_a3_v3s2(
V3_S2* v0, V3_S2* v1, V3_S2* v2,
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2,
A2_S2* pp, S4* otz, S4* flag
){
return rtp_avg_nclip_a3_v3s2_raw(
v0, v1, v2,
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
);
}
S4 rtp_avg_nclip_a4_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3, S4* xy1, S4* xy2, S4* xy3, S4* xy4, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip4");
FI_ S4 rtp_avg_nclip_a4_v3s2(
V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3,
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2, V2_S2* xy3,
A2_S2* pp, S4* otz, S4* flag
){
return rtp_avg_nclip_a4_v3s2_raw(
v0, v1, v2, v3,
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
);
}
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
+14 -1
View File
@@ -6,7 +6,7 @@
// One line per macro that appears in your atom sources. // One line per macro that appears in your atom sources.
// //
// This file is encoding-macros-only. // This file is encoding-macros-only.
// The auto-generated component macros (mac_X) live in duffle/gen/<dir>.macs.h (included separately by the unity build). // The auto-generated component macros (mac_X) live in the source directory's own gen/macs.h (per-directory aggregation; included separately by the unity build).
// The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped // The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped
// (or the include guard order handled) to avoid WORD_COUNT redeclaration. // (or the include guard order handled) to avoid WORD_COUNT redeclaration.
// //
@@ -22,6 +22,7 @@ WORD_COUNT(call_reg, 1)
WORD_COUNT(call_addr, 1) WORD_COUNT(call_addr, 1)
WORD_COUNT(branch_le_zero, 1) WORD_COUNT(branch_le_zero, 1)
WORD_COUNT(branch_equal, 1) WORD_COUNT(branch_equal, 1)
WORD_COUNT(branch_ne, 1)
WORD_COUNT(add_ui, 1) WORD_COUNT(add_ui, 1)
WORD_COUNT(set_lt_u, 1) WORD_COUNT(set_lt_u, 1)
WORD_COUNT(set_lt_s, 1) WORD_COUNT(set_lt_s, 1)
@@ -29,7 +30,9 @@ WORD_COUNT(set_lt_si, 1)
WORD_COUNT(set_lt_ui, 1) WORD_COUNT(set_lt_ui, 1)
WORD_COUNT(load_word, 1) WORD_COUNT(load_word, 1)
WORD_COUNT(load_half_u, 1) WORD_COUNT(load_half_u, 1)
WORD_COUNT(load_byte_u, 1)
WORD_COUNT(store_word, 1) WORD_COUNT(store_word, 1)
WORD_COUNT(store_byte, 1)
WORD_COUNT(add_ui_self, 1) WORD_COUNT(add_ui_self, 1)
WORD_COUNT(add_u_self, 1) WORD_COUNT(add_u_self, 1)
WORD_COUNT(add_u, 1) WORD_COUNT(add_u, 1)
@@ -37,6 +40,7 @@ WORD_COUNT(or_i, 1)
WORD_COUNT(or_i_self, 1) WORD_COUNT(or_i_self, 1)
WORD_COUNT(or_u, 1) WORD_COUNT(or_u, 1)
WORD_COUNT(or_u_self, 1) WORD_COUNT(or_u_self, 1)
WORD_COUNT(nor_u, 1)
WORD_COUNT(shift_lleft, 1) WORD_COUNT(shift_lleft, 1)
WORD_COUNT(shift_lleft_self, 1) WORD_COUNT(shift_lleft_self, 1)
WORD_COUNT(shift_lright, 1) WORD_COUNT(shift_lright, 1)
@@ -50,6 +54,15 @@ WORD_COUNT(gte_sw, 1)
WORD_COUNT(gte_cmdw_rtpt, 1) WORD_COUNT(gte_cmdw_rtpt, 1)
WORD_COUNT(gte_cmdw_nclip, 1) WORD_COUNT(gte_cmdw_nclip, 1)
WORD_COUNT(gte_avg_sort_z3, 1) WORD_COUNT(gte_avg_sort_z3, 1)
WORD_COUNT(gte_cmdw_sqr, 1)
WORD_COUNT(gte_cmdw_gpf, 1)
WORD_COUNT(shift_lleft_var, 1)
WORD_COUNT(shift_aright_var, 1)
WORD_COUNT(li_s, 1)
WORD_COUNT(and_i, 1)
WORD_COUNT(add_si, 1)
WORD_COUNT(branch_lt_zero, 1)
WORD_COUNT(sub_s, 1)
WORD_COUNT(sub_u, 1) WORD_COUNT(sub_u, 1)
WORD_COUNT(nop2, 2) WORD_COUNT(nop2, 2)
-156
View File
@@ -1,156 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "duffle/gen/duffle.macs.h"
# include "duffle/gen/duffle.offsets.h"
# include "duffle/atom_dsl.h"
# include "duffle/lottes_tape.h"
# include "duffle/word_count.metadata.h"
# include "gen/gte_hello.offsets.h"
# include "hello_gte.h"
#endif
#pragma region MACs (Mips Atom components)
#pragma endregion MACs
#pragma region Baked Atoms
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple,
nop2, gte_cmdw_nclip,
nop2, gte_mv_from_data_r(R_T0, C2_MAC0),
nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
mac_format_g4_color(
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
mac_gte_store_g4_p012_post_rtpt_pre_rtps(),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
nop2, gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3_post_rtps(),
nop2, gte_cmdw_avg_sort_z4,
nop2, gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(),
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursr)
) {
mac_load_tri_indices( R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_gte_store_f3_post_rtpt(),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
/* Insert into Ordering Table Linked List */
mac_insert_ot_tag_f3(),
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
#pragma endregion Baked Atoms
+13
View File
@@ -0,0 +1,13 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
// Per-phase register allocations resolved by the lua pass.
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
#define R_GpTmp_Code R_V0_Code
+41
View File
@@ -0,0 +1,41 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_camera/
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
#ifndef WORD_COUNT
#define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif
#define mac_put_disp_env(reg_transfer, reg_base, port) \
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
WORD_COUNT(mac_put_disp_env, 5)
#define mac_put_draw_env(reg_transfer, reg_base, port) \
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
WORD_COUNT(mac_put_draw_env, 16)
+68
View File
@@ -0,0 +1,68 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_camera\
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
#pragma once
#pragma region hello_camera
// --- atom: pad_input_cube_rotation (60 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6
#define _atom_offset_dead_zone_low_check_dead_low_active 8
#define _atom_offset_dead_zone_high_check_dead_high_active 15
#define _atom_offset_dead_zone_skip_exit_stick 24
#define _atom_offset_end_low_exit_stick 12
enum {
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
};
// --- atom: pad_input_cam (40 words) ---
#define _atom_offset_left_x_exit_left_x 3
#define _atom_offset_right_x_exit_right_x 3
#define _atom_offset_up_y_exit_up_y 3
#define _atom_offset_down_y_exit_down_y 3
#define _atom_offset_cross_z_exit_cross_z 3
#define _atom_offset_circle_z_exit_circle_z 3
enum {
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
};
// --- atom: cube_g4_face (76 words) ---
#define _atom_offset_cull_cube_g4_face_exit 41
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
};
// --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
};
#pragma endregion hello_camera
+924
View File
@@ -0,0 +1,924 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "duffle/gen/macs.h"
# include "duffle/gen/offsets.h"
# include "duffle/dsl.atom.h"
# include "duffle/lottes_tape.h"
# include "duffle/mips.h"
# include "duffle/gte.h"
# include "duffle/gp.h"
# include "duffle/pad.h"
# include "duffle/word_count.metadata.h"
# include "duffle/psyq.h"
# include "duffle/math.atom.c"
# include "duffle/mips.atom.c"
# include "duffle/gte.atom.c"
# include "duffle/gp.atom.c"
# include "duffle/psyq.atom.c"
# include "gen/offsets.h"
# include "gen/macs.h"
# include "gen/auto_reg.h"
# include "hello_camera.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
/*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References:
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
*
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
* ./toolchain/psyq-4_7/lib/libgpu.a
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
*
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
* code[8..10] = padding (NOP) — 3 words to fill the packet
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
* code[13..14] = padding (NOP) — completes the 16-word packet
*/
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
})
#pragma endregion MACs
#pragma region Atom Procs
// Modular Atoms
/* Scratchpad layout for the resolve_look_at bundle.
* The chain atoms communicate entirely via the wave-context GPR carrier R_ResolveScratch (R_T4) + hardcoded offsets into smem.scratchpad
* (PS1 hardware scratchpad at 0x1F800000).
*
* Atom 0 (input_and_sub) STAGES the C-side inputs (eye, up_in) into the scratchpad;
* AT THE SAME TIME it computes fwd = target - eye and stores it at scratch+0.
* Atoms 1-6 then read/write specific scratchpad offsets internally using
* `r_scratch + hardcoded_offset` — no tape-data pointers are passed between atoms.
* +0 fwd (atom 0 writes; atom 1 reads)
* +16 uz (atom 1 writes; atoms 2 + 4 read)
* +32 right (atom 2 writes; atom 3 reads)
* +48 ux (atom 3 writes; atoms 4 + 6 read)
* +64 up (atom 4 writes; atom 5 reads)
* +80 uy (atom 5 writes; atom 6 reads)
* +96 eye (atom 0 stages from C-side pointer; atom 6 reads)
* +128 up_in (atom 0 stages from C-side pointer; atom 2 reads)
*/
// enum {
// R_LookAt = R_T0 atom_reg atom_type(MT3_S2S4*),
// R_CamEye = R_T1 atom_reg atom_type(P3_S4*),
// R_CamTarget = R_T2 atom_reg atom_type(P3_S4*),
// R_WorldUp = R_T3 atom_reg atom_type(V3_S4*),
// };
enum {
/* Wave-context GPR carrier for the resolve_look_at bundle: the scratch base.
* Set by atom 0 (popped from tape), read by atoms 1-6 (used as pointer base). */
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
};
typedef Struct_(Binds_ResolveLookAt) {
MT3_S2S4* look_at;
P3_S4* eye;
P3_S4* target;
V3_S4* up_in;
};
/* Per-atom bind-pop structs for the resolve_look_at bundle. */
typedef Struct_(Binds_ResolveLookAtScratch) {
U4 scratch_base; /* U4 (scratch base address — populated by helper with u4_(smem.scratchpad)) */
};
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's
* scratchpad slots (PS1 hardware scratchpad at 0x1F800000).
*
* Each slot is 16 bytes: V3_S4 is already 16 bytes (4 × S4 = x/y/z/pad).
* The struct fields are contiguous — slot i starts at offset i*16.
* Used by the assembly via O_(ResolveLookAtScratch, fld.x/y/z) which resolves to a compile-time byte offset.
* NOT a runtime struct — the struct is purely a schema for offsets; the assembly uses `r_scratch + O_(...)` to compute slot addresses at runtime.
*
* Slot producers/consumers (referenced by the resolve_look_at chain atoms):
* +0 fwd 0 writes (target - eye); atom 1 (normalize) reads
* +16 uz 1 writes (normalize fwd); atoms 2 + 4 read (cross operands)
* +32 right 2 writes (cross uz x up_in); atom 3 (normalize) reads
* +48 ux 3 writes (normalize right); atoms 4 + 6 read
* +64 up 4 writes (cross uz x ux); atom 5 (normalize) reads
* +80 uy 5 writes (normalize up); atom 6 reads
* +96 eye 0 stages (C-side input); atom 6 reads (translation column)
* +112 target reserved (currently written nowhere — kept for symmetry w/ eye)
* +128 up_in 0 stages (C-side input); atom 2 reads (cross operand)
*
* Fields use P3_S4 (point) for eye/target (RGA: affine point, implicit weight 1);
* V3_S4 (vector) for fwd/uz/right/ux/up/uy/up_in (RGA: Euclidean vector).
* P3_S4 is a storage alias of V3_S4 (see math.h comment: "Storage alias of V3_S4.
* Use P3_S4 when the value is a point.") — both are 16 bytes.
*/
typedef Struct_(ResolveLookAtScratch) {
V3_S4 fwd; /* offset +0 (16 bytes — 4 S4 fields incl. internal pad) */
V3_S4 uz; /* offset +16 (16 bytes) */
V3_S4 right; /* offset +32 (16 bytes) */
V3_S4 ux; /* offset +48 (16 bytes) */
V3_S4 up; /* offset +64 (16 bytes) */
V3_S4 uy; /* offset +80 (16 bytes) */
P3_S4 eye; /* offset +96 (16 bytes; storage alias of V3_S4) */
P3_S4 target; /* offset +112 (16 bytes; storage alias of V3_S4) */
V3_S4 up_in; /* offset +128 (16 bytes) */
};
/* ─── resolve_look_at bundle chain atoms ────────────────────────────
* 4 unique atom procs in the resolve_look_at bundle (4 chain atoms + 3 calls to generic normalize_v3s4_proc).
* All 4 chain atoms are runtime-built MipsAtom_Proc_ atoms: each function declares a static MipsCode[] body,
* then calls atombuilder_unroll() to append it to the caller's MipsAtomBuilder arena. resolve_look_at_init()
* uses this pattern to pre-build the bundle into the static arena (smem.resolve_look_at_arena).
*
* Atom roster:
* 0: resolve_look_at__input_and_sub (chain atom)
* 1: normalize_v3s4_proc (gte.atom.c) (generic normalize; called for fwd→uz)
* 2: resolve_look_at__cross_uz_up_in_to_right (chain atom)
* 3: normalize_v3s4_proc (gte.atom.c) (generic normalize; called for right→ux)
* 4: resolve_look_at__cross_uz_ux_to_up (chain atom)
* 5: normalize_v3s4_proc (gte.atom.c) (generic normalize; called for up→uy)
* 6: resolve_look_at__populate_and_translate (chain atom)
*
* The generic normalize_v3s4_proc is a parameterized 4-stage GTE normalize (SQR → mfc2 → LZCS → GPF → srav);
* it accepts scratch base + offset args so any caller (with a scratch base + struct schema) can use it.
*/
typedef Struct_(Binds_ResolveLookAtSub) {
U4 target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
U4 eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
U4 up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
};
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
* Inputs (C-side pointers popped from the tape):
* r_target_ptr : P3_S4* (C-side struct; atom 0 reads target.x/y/z directly)
* r_eye_ptr : P3_S4* (C-side struct; staged into scratchpad at +96/+100/+104)
* r_up_in_ptr : V3_S4* (C-side struct; staged into scratchpad at +128/+132/+136)
* Wave-context output:
* r_scratch : R_ResolveScratch (R_T4) — scratch base, read by atoms 1-6
*
* Bind-pop layout:
* Binds_ResolveLookAtSub = 12 bytes (target + eye + up_in ptrs)
* Binds_ResolveLookAtScratch = 4 bytes (scratch_base)
* Staging work:
* * Stage eye.x/y/z → scratch+96/+100/+104 (for atom 6's translation column)
* * Stage up_in.x/y/z → scratch+128/+132/+136 (for atom 2's outer-product operand)
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
*
* GPR codes (assigned by resolve_look_at_init):
* r_target_ptr : R_T0
* r_eye_ptr : R_T1
* r_up_in_ptr : R_T2
* r_scratch : R_T4 (R_ResolveScratch; wave-context carrier)
* r_tmp0 : R_T3 (stage eye/up_in + load eye.y)
* r_tmp1 : R_T5 (stage eye/up_in + load eye.z)
* r_tmp2 : R_T6 (stage eye/up_in + load target.x)
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
* R_AT : hardcoded (load eye.y / eye.z / target.z)
* R_V0 : hardcoded (load eye.z / target.z)
*
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
*/
I_ void resolve_look_at__input_and_sub_proc(MipsAtomBuilder_R ab, U4 r_scratch
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
) MipsAtom_Proc_(resolve_look_at__input_and_sub, ab, {
/* Pop the 3 C-side pointers + scratch_base from the tape. */
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtScratch,scratch_base)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtScratch)),
/* Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation
* column). Reuse r_tmp0/r_tmp1/r_tmp2. Offsets via O_(ResolveLookAtScratch,*). */
load_word(r_tmp0, r_eye_ptr, O_(P3_S4,x)),
load_word(r_tmp1, r_eye_ptr, O_(P3_S4,y)),
load_word(r_tmp2, r_eye_ptr, O_(P3_S4,z)),
nop, /* load-delay */
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,eye.x)),
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,eye.y)),
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye.z)),
/* Stage up_in.x/y/z into the scratchpad (atom 2 reads these for the outer
* product with uz). Reuse r_tmp0/r_tmp1/r_tmp2. */
load_word(r_tmp0, r_up_in_ptr, O_(V3_S4,x)),
load_word(r_tmp1, r_up_in_ptr, O_(V3_S4,y)),
load_word(r_tmp2, r_up_in_ptr, O_(V3_S4,z)),
nop, /* load-delay */
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,up_in.x)),
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,up_in.y)),
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in.z)),
/* Compute fwd = target - eye. */
load_word(r_tmp0, r_target_ptr, O_(P3_S4,x)),
load_word(r_tmp1, r_target_ptr, O_(P3_S4,y)),
load_word(r_tmp2, r_target_ptr, O_(P3_S4,z)),
load_word(r_tmp3, r_eye_ptr, O_(P3_S4,x)),
load_word(R_AT, r_eye_ptr, O_(P3_S4,y)),
load_word(R_V0, r_eye_ptr, O_(P3_S4,z)),
nop, /* load-delay */
sub_u(r_tmp0, r_tmp0, r_tmp3),
sub_u(r_tmp1, r_tmp1, R_AT),
sub_u(r_tmp2, r_tmp2, R_V0),
/* Store fwd.x/y/z (atom 1 reads these as the normalize src). */
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,fwd.x)),
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,fwd.y)),
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd.z)),
mac_yield()
})
/* Atoms 2 + 4 in the bundle: out = a × b (GTE outer product on IR/D vectors).
* No bind pop — the three operand pointers (a, b, out) are derived in-body from r_scratch + hardcoded_offset.
* Each atom has its own variant because the offsets are baked into the body and each atom uses unique GPRs.
*
* GTE register layout (per PSX-SPX + duffle gte.h):
* IR1/2/3 = a.x/y/z (mtc2)
* VXY0 = b.x (mtc2)
* VZ0 = b.y (mtc2)
* VXY1 = b.z (mtc2)
* OP = outer product
* MAC1/2/3 = out.x/y/z (mfc2)
*
* Pool cost: r_scratch (R_T4 carrier) + 7 body GPRs + R_AT + R_V0 (hardcoded) = 10 GPRs.
*/
/* Atom 2: cross uz × up_in → right. */
I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, ab, {
/* Compute the three scratch pointers from r_scratch. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0
(hardcoded; reusing the body's last two loads is fine because the load-delay slot is the nop after the third load,
and mtc2 below doesn't read these regs). */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */
gte_mv_to_data_r(r_a, C2_IR1),
gte_mv_to_data_r(r_b, C2_IR2),
gte_mv_to_data_r(r_c, C2_IR3),
gte_mv_to_data_r(r_d, C2_VXY0), /* D1 = b.x */
gte_mv_to_data_r(R_AT, C2_VZ0), /* D2 = b.y */
gte_mv_to_data_r(R_V0, C2_VXY1), /* D3 = b.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product, /* OP fires; MAC1/2/3 = a × b */
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop, /* MFC2 retirement */
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_yield()
})
/* Atom 4: cross uz × ux → up. */
I_ void resolve_look_at__cross_uz_ux_to_up_proc(MipsAtomBuilder_R ab, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, ab, {
/* Compute the three scratch pointers from r_scratch. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */
gte_mv_to_data_r(r_a, C2_IR1),
gte_mv_to_data_r(r_b, C2_IR2),
gte_mv_to_data_r(r_c, C2_IR3),
gte_mv_to_data_r(r_d, C2_VXY0),
gte_mv_to_data_r(R_AT, C2_VZ0),
gte_mv_to_data_r(R_V0, C2_VXY1),
nop2,
gte_cmdw_outer_product,
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop,
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_yield()
})
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
};
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye).
*
* GPR codes (assigned by resolve_look_at_init):
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
*
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
*
* Struct layout (per duffle/math.h):
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
*
* Translation column: GTE MVMVA with the world rotation matrix pre-set
* (helper emits set_gte_world before the bundle, per the bundle design).
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
*/
I_ void resolve_look_at__populate_and_translate_proc(MipsAtomBuilder_R ab
, U4 r_look_at
, U4 r_scratch
, U4 r_pux, U4 r_puy, U4 r_puz, U4 r_peye /* 4 dedicated pointer regs */
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2 /* 3 atom-local scratch regs */
) MipsAtom_Proc_(resolve_look_at__populate_and_translate, ab, {
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* Compute the 4 scratch pointers in their dedicated GPRs. */
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)), /* r_peye = &eye */
nop,
/* ── m[0] = (S2)ux ── */
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
/* ── m[1] = (S2)uy ── */
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
/* ── m[2] = (S2)uz ── */
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
/* ── Translation column t[i] = R * (-eye) ─────────────────────────────
* pos = -eye: load eye.x/y/z from r_peye, negate via sub_u from R_0. */
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
nop,
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
sub_u(r_tmp1, R_0, r_tmp1),
sub_u(r_tmp2, R_0, r_tmp2),
/* mtc2 IR1/2/3 = pos (for MVMVA — input vector registers). */
gte_mv_to_data_r(r_tmp0, C2_IR1),
gte_mv_to_data_r(r_tmp1, C2_IR2),
gte_mv_to_data_r(r_tmp2, C2_IR3),
nop2,
/* MVMVA: MAC1/2/3 = R * IR with cv=0 (no TR vector), mx=0 (rotation matrix), sf=0 (no shift), v=0 (V0 = IR1/2/3, no far-plane clipping).
* The pre-set rotation matrix is the one set by the preceding set_gte_world atom.
* gte_cmdw_mvmva is parameterless and defaults to cv=0/mx=0/sf=0/v=0. */
gte_cmdw_mvmva,
nop, /* GTE interlock */
/* mfc2 MAC1/2/3 → r_tmp0/r_tmp1/r_tmp2 (sign-extended into 32-bit GPRs).
* MAC1/2/3 hold R*v with no TR add and no perspective divide — exactly the 3 distinct world-space translation values we need for t[0..2]. */
gte_mv_from_data_r(r_tmp0, C2_MAC1),
gte_mv_from_data_r(r_tmp1, C2_MAC2),
gte_mv_from_data_r(r_tmp2, C2_MAC3),
nop,
store_word(r_tmp0, r_look_at, O_(MT3_S2S4,t[0])),
store_word(r_tmp1, r_look_at, O_(MT3_S2S4,t[1])),
store_word(r_tmp2, r_look_at, O_(MT3_S2S4,t[2])),
mac_yield()
})
#pragma endregion Atom Procs
#pragma region Baked Atoms
enum {
R_ScreenX = R_T5 atom_reg atom_type(U2),
R_ScreenY = R_T6 atom_reg atom_type(U2),
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
#define R_ScreenBuf_Code R_T7_Code
};
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
) {
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
add_ui(R_T0, R_0, gp0_tpage_default),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
add_ui(R_T0, R_0, 1),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
add_ui(R_T0, R_0, 7),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
mac_yield(),
};
/* gp_screen_init's GPR setup. Tests the mixed user-pinning + auto-reg pattern:
* - R_IO_BaseAddr = R_T4 (user-pinned via atom_reg; pre-existing)
* - R_GP1_Offset = R_T2 (user-pinned via atom_reg; NEW -- for GPIO_PORT1_OFFSET)
* - R_ScreenX = R_T5 (user-pinned via atom_reg; used as a transfer and GTE setup reg)
* - R_GpTmp = auto-allocated by the lua pass and used for several GPU transfers;
* the C preprocessor resolves it to the chosen free pool GPR.
*
* For gp_screen_init, the auto-reg pool exclusions are:
* user_pinned (from the corpus register_alias_registry) : R_T0..R_T7 (all 8 user-pinned across hello_camera.atom.c)
* body-parsed physical registers : aliases resolve through the registry;
* the body uses R_ScreenX, not raw R_T5
* source_pool after both subtractions : {R_V0, R_V1} only
* R_GpTmp gets R_V0 (the first-fit choice). Its repeated GPU-transfer use proves that the
* auto-reg allocation is active while the R_ScreenX references prove the pinned alias is used.
* R_TapePtr (R_T9), R_AtomJmp (R_T8), R_AT are excluded from the POOL by construction in
* passes/auto_reg.lua -- see the "obvious exclusions" comment block at the top of that file.
*/
enum {
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
atom_auto_reg(gp_screen_init, R_GpTmp), /* Auto-allocated scratch; resolved to a free pool GPR by the lua pass. C-preprocessor expands to R_GpTmp = R_GpTmp_Code with an atom_auto_reg trailing comment. */
#define R_IO_BaseAddr_Code R_T4_Code
#define R_GP1_Offset_Code R_T2_Code
};
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO; uses pinned R_ScreenX as the transfer reg. */
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ; uses pinned R_ScreenX as the transfer reg. */
mac_gcmd_push(gp1_word_DisplayOn(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON; uses pinned R_ScreenX as the transfer reg. */
mac_gcmd_push(gp1_word_dma_to_gpu(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU->GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. Uses auto-allocated R_GpTmp. */
mac_gcmd_push(gp1_word_StartDisplayArea(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0); uses auto-allocated R_GpTmp. */
/* GP1: DisplayMode + Display Ranges. */
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
load_upper_i(R_ScreenX, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFX_Code),
load_upper_i(R_ScreenX, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFY_Code),
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
add_ui(R_ScreenX, R_0, ScreenZ), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_H_Code),
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
mac_gcmd_push(gp1_word_DisplayOn(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* Uses auto-allocated R_GpTmp. */
mac_yield(),
};
typedef Struct_(Binds_PadApplyInput) {
PadState* state;
V3_S2* cube_rot;
V3_S2* floor_rot;
};
enum {
R_PadStateT5 = R_T5 atom_reg,
R_CubeRot = R_T1 atom_reg,
R_FloorRot = R_T2 atom_reg,
};
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
, atom_writes( R_CubeRot, R_FloorRot)
) {
/* Pop Binds from tape (state, cube_rot, floor_rot) */
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
/* Load pad[0].buttons into R_T0. */
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
// Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30),
add_si( R_T3, R_T3, 5),
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30),
add_si( R_T3, R_T3, -5),
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_dpad_right)
/* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
add_ui( R_T4, R_0, PadDeadZone_HighBound),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
mac_yield_load(),
atom_label(dead_low_active)
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
/* delta = 0x80 - left_x (positive). */
/* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
jump_rel(atom_offset(end_low, exit_stick)),
mac_yield_load(),
atom_label(dead_high_active)
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
sub_u( R_T3, R_T4, R_T3),
/* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
atom_label(no_jump_fallthrough)
mac_yield_load(),
atom_label(exit_stick)
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
mac_yield_tail(),
};
enum {
R_Cam = R_T4 atom_reg,
R_CamPadState = R_T5 atom_reg,
};
typedef Struct_(Binds_PadInputCam) {
PadState* state;
Camera* cam;
};
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
, atom_writes(R_Cam)
) {
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_left_x)
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_right_x)
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_up_y)
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_down_y)
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_cross_z)
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_circle_z)
mac_yield_tail(),
};
enum {
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* Output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
#define R_PrimCursor_Code R_T7_Code
#define R_FaceCursor_Code R_T4_Code
#define R_VertBase_Code R_T5_Code
#define R_OtBase_Code R_T6_Code
};
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(R_PrimCursor),
gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3(R_PrimCursor),
gte_cmdw_avg_sort_z4,
gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
mac_format_g4_color(R_PrimCursor,
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
// end: branch(bounds_chk)
// end: branch(cull)
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
// atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(R_PrimCursor),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk)
// end: branch(culling)
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
#pragma endregion Baked Atoms
+546
View File
@@ -0,0 +1,546 @@
#pragma region Vendors
#include <stdio.h>
#include <stdlib.h>
#include <assert.h>
// #include "libgpu.h"
// #include "libetc.h"
// #include "libgte.h"
#pragma endregion Vendors
#pragma region Duffle Headers
# include "duffle/gen/macs.h"
# include "duffle/gen/offsets.h"
#include "duffle/word_count.metadata.h"
#include "duffle/dsl.h"
#include "duffle/memory.h"
#include "duffle/math.h"
#include "duffle/gcc_asm.h"
#include "duffle/mips.h"
#include "duffle/gp.h"
#include "duffle/gte.h"
#include "duffle/pad.h"
#include "duffle/dsl.atom.h"
#include "duffle/lottes_tape.h"
#include "duffle/bios.h"
#include "duffle/psyq.h"
#pragma endregion Duffle Headers
#pragma region Duffle TUs
#include "duffle/pad.c"
#include "duffle/math.atom.c"
#include "duffle/mips.atom.c"
#include "duffle/gte.atom.c"
#include "duffle/gp.atom.c"
#include "duffle/pad.atom.c"
#include "duffle/psyq.atom.c"
#pragma endregion Duffle TUs
#pragma region Hello Camera Headers
# include "gen/macs.h"
# include "gen/offsets.h"
# include "gen/auto_reg.h"
#include "hello_camera.h"
#pragma endregion Hello Camera Headers
#pragma region Hello Joypad TUs
#include "hello_camera.atom.c"
#pragma endregion Hello Joypad TUs
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
ResolveLookAtArena_Words = 512,
};
typedef Struct_(SMemory) {
PrimitiveArena primitives;
A2_OrderingTable_Buffer ordering_tbl;
DoubleBuffer screen_buf;
S4 active_buf_id;
U4 MemTape[MemTape_Len];
MT3_S2S4 tform_world;
MT3_S2S4 tform_view;
Camera cam;
Ent_Cube cube;
Ent_Floor floor;
PadBiosRaw pad_raw[2];
PadState pad[2];
U4_V scratchpad; // d-cache
/* resolve_look_at bundle: pre-built atom arena + atom-refs.
* (Task 12.5 fix: moved from file-scope globals to smem fields.
* Task 12.7 fix: dropped the ResolveLookAtScratch struct-as-view; the
* C-side helper uses `& smem.scratchpad[N]` at hardcoded offsets directly.
* Task 12.8 fix: chain atoms use r_scratch + offset internally; no C-side magic offsets anywhere.
* Task 12.11 fix: ResolveLookAtScratch offset schema moved to hello_camera.atom.c — gte.atom.c
* is the GENERIC GTE primitives file and must not know about the resolve_look_at bundle's scratch layout.) */
U4 resolve_look_at_arena[ResolveLookAtArena_Words]; /* ~2 KB; bumped from 420 per Task 4 subagent */
MipsAtom* resolve_look_at_atom_addrs[7];
MipsAtomBuilder resolve_look_at_ab_static;
};
global SMemory smem;
extern SMemory smem;
#define pad0_btn_(btn) btn & smem.pad[0].buttons
#define pad1_btn_(btn) btn & smem.pad[1].buttons
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
gknown PrimitiveArena* pa = & smem.primitives;
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
assert(pa->used + type_width < PrimitiveBuff_Len);
B1* next = buf + pa->used;
pa->used += type_width;
return next;
}
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
void
resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
// Preconditions: eye != target, up_in not collinear with (target - eye).
V3_S4 right, up, forward;
V3_S4 ux, uy, uz;
V3_S4 pos, off;
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
mul_m3s2_v3s4(look_at, & pos, & off);
trans_m3s2( look_at, & off);
}
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
* Called ONCE from main() before the frame loop.
* After this returns, the smem.resolve_look_at_atom_addrs[] array contains valid MIPS atom pointers
* for the frame-time bundle helper to emit via tb_emit(tb, captured_addr).
*
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
* share the GENERIC normalize_v3s4_proc from gte.atom.c (called 3x with different
* O_(ResolveLookAtScratch,...) offsets):
* 0: resolve_look_at__input_and_sub_proc
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
* 4: resolve_look_at__cross_uz_ux_to_up_proc
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
* 6: resolve_look_at__populate_and_translate_proc
*
* Task 12.16 promotion: the bundle-specific resolve_look_at__chain_normalize_proc
* has been promoted to the generic normalize_v3s4_proc (gte.atom.c), which now
* takes r_scratch + r_src_offset + r_dst_offset as U4 parameters. The 3 callers
* pass O_(ResolveLookAtScratch,...) macros as offset args. The metaprogram emits
* one set of `atom_offset__normalize_v3s4__srav_path__aligned_done` defs
* (namespaced by atom name) in duffle/gen/offsets.h, shared by all 3 callers.
*
* GPR pool per atom: 10 free GPRs (R_T0..R_T3 + R_T5..R_T7 + R_V0 + R_V1 + R_AT).
* R_T4 is reserved as the wave-context carrier (R_ResolveScratch).
*/
internal void resolve_look_at_init(void) {
/* Wrap the static arena in a MipsAtomBuilder. */
MipsAtomBuilder_R ab = & smem.resolve_look_at_ab_static;
ab->start = u4_(smem.resolve_look_at_arena);
ab->capacity = ResolveLookAtArena_Words;
ab->used = 0;
/* Atom 0: resolve_look_at__input_and_sub — stages eye/up_in into scratchpad,
* computes fwd = target - eye; binds R_ResolveScratch (R_T4) as the wave-context carrier for atoms 1-6.
* The body hardcodes R_AT and R_V0 as eye.y/eye.z temps (the existing sub_u(eye.x, eye.y, eye.z) chain from the prior Task 12.7 design). */
smem.resolve_look_at_atom_addrs[0] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__input_and_sub_proc(ab, R_ResolveScratch,
R_T0, /* r_target_ptr (popped from tape) */
R_T1, /* r_eye_ptr (popped from tape) */
R_T2, /* r_up_in_ptr (popped from tape) */
R_T3, R_T5, R_T6, R_T7); /* r_tmp<0-3> */
/* Atom 1: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+0=fwd, dst=scratch+16=uz.
* The proc takes r_src_offset + r_dst_offset as U4 PARAMETERS — we pass the O_(...) macros here (evaluating to numeric literals 0 and 16).
* The 4-stage body is identical across the 3 call sites (atoms 1, 3, 5); only the offset args differ.
* GPR pool: r_scratch (R_T4 carrier) + 9 body GPRs = 10.
* r_src_ptr (R_T0) : src ptr
* r_dst_ptr (R_T1) : dst ptr
* r_tmp (R_T2) : unused (reserved for symmetry)
* r_mac1_scratch (R_T3) : MAC1 scratch
* r_mac2_scratch (R_T5) : src.x → result.x (carries through stages 1-2)
* r_recip_est (R_T6) : src.y → result.y
* r_lzcr (R_T7) : |v|² accumulator + srav amount (single reg)
* r_shift (R_V0) : LZCR (saved across stages 3-4)
* r_branch_tmp (R_V1) : src.z → result.z (reused after stage 1)
*/
smem.resolve_look_at_atom_addrs[1] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
normalize_v3s4_proc(ab, R_ResolveScratch, /* r_scratch (wave-context carrier) */
O_(ResolveLookAtScratch, fwd), /* r_src_offset = 0 */
O_(ResolveLookAtScratch, uz), /* r_dst_offset = 16 */
R_T0, R_T1, R_T2, /* r_src_ptr, r_dst_ptr, r_tmp */
R_T3, /* r_mac1_scratch */
R_T5, /* r_mac2_scratch */
R_T6, /* r_recip_est */
R_T7, /* r_lzcr */
R_V0, /* r_shift */
R_V1); /* r_branch_tmp */
/* Atom 2: resolve_look_at__cross_uz_up_in_to_right — a=scratch+16, b=scratch+128,
* out=scratch+32 (HARDCODED in body). GPR pool: r_scratch + 7 body + R_AT + R_V0 = 10. */
smem.resolve_look_at_atom_addrs[2] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__cross_uz_up_in_to_right_proc(ab, R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */
R_T0, R_T1, R_T2, /* r_a, r_b, r_c (a.x/y/z → out.x/y/z) */
R_T3, /* r_d (b.x) */
R_T5, /* r_f (out ptr = scratch+32) */
R_T6, /* r_g (a ptr = scratch+16) */
R_T7); /* r_h (b ptr = scratch+128) */
/* Atom 3: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+32=right, dst=scratch+48=ux. */
smem.resolve_look_at_atom_addrs[3] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
normalize_v3s4_proc(ab, R_ResolveScratch,
O_(ResolveLookAtScratch, right), /* r_src_offset = 32 */
O_(ResolveLookAtScratch, ux), /* r_dst_offset = 48 */
R_T0, R_T1, R_T2,
R_T3,
R_T5,
R_T6,
R_T7,
R_V0,
R_V1);
/* Atom 4: resolve_look_at__cross_uz_ux_to_up — a=scratch+16, b=scratch+48, out=scratch+64 (HARDCODED). */
smem.resolve_look_at_atom_addrs[4] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__cross_uz_ux_to_up_proc(ab, R_ResolveScratch,
R_T0, R_T1, R_T2,
R_T3,
R_T5, /* r_f (out ptr = scratch+64) */
R_T6, /* r_g (a ptr = scratch+16) */
R_T7); /* r_h (b ptr = scratch+48) */
/* Atom 5: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+64=up, dst=scratch+80=uy. */
smem.resolve_look_at_atom_addrs[5] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
normalize_v3s4_proc(ab, R_ResolveScratch,
O_(ResolveLookAtScratch, up), /* r_src_offset = 64 */
O_(ResolveLookAtScratch, uy), /* r_dst_offset = 80 */
R_T0, R_T1, R_T2,
R_T3,
R_T5,
R_T6,
R_T7,
R_V0,
R_V1);
/* Atom 6: resolve_look_at__populate_and_translate — write look_at->m[][] from ux/uy/uz (computed from r_scratch+offset internally),
then compute translation column t[] = R * (-eye). GPR pool: r_look_at + r_scratch + 4 ptr regs + 3 tmp regs = 9. */
smem.resolve_look_at_atom_addrs[6] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__populate_and_translate_proc(ab,
R_T0, /* r_look_at (popped from tape; MT3_S2S4*) */
R_ResolveScratch, /* r_scratch (wave-context carrier) */
R_T1, R_T3, R_T5, R_T7, /* r_pux, r_puy, r_puz, r_peye */
R_T2, R_T6, R_V0); /* r_tmp0, r_tmp1, r_tmp2 */
/* Sanity check: arena didn't overflow. */
assert(ab->used <= ResolveLookAtArena_Words);
}
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
* The 7 chain atoms are pre-built at init time (resolve_look_at_init) and referenced by address via smem.resolve_look_at_atom_addrs[].
* Per-frame work: 7 tb_emit (atom pointer emissions) + 5 tb_data (C-side pointers for atom 0 + look_at for atom 6).
*
* Binds_ contract (the field-name labels are for human readability):
* Atom 0 input_and_sub target(4) eye(4) up_in(4) scratch_base(4) = 4 words
* Atoms 1-5 (no tape data — atom uses r_scratch + offset internally)
* Atom 6 populate_and_translate look_at(4) = 1 word
* ----
* 5 tb_data words total per frame.
*/
I_ void resolve_look_at(
TapeBuilder_R tb
, MT3_S2S4* look_at
, P3_S4* eye
, P3_S4* target
, V3_S4* up_in
){
/* Atom 0: input_and_sub — stages eye/up_in into scratchpad + computes fwd. */
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
tb_data(tb, u4_(target)); /* Binds_ResolveLookAtSub.target (C-side P3_S4*) */
tb_data(tb, u4_(eye)); /* Binds_ResolveLookAtSub.eye (C-side P3_S4*) */
tb_data(tb, u4_(up_in)); /* Binds_ResolveLookAtSub.up_in (C-side V3_S4*) */
tb_data(tb, u4_(smem.scratchpad)); /* Binds_ResolveLookAtScratch.scratch_base */
}
/* Atoms 1-5: NO tb_data — each chain atom uses r_scratch + hardcoded_offset internally (no tape-data pointers between atoms).
Context carrier R_ResolveScratch (R_T4) is preserved across atoms. */
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
/* Atom 6: populate_and_translate — only output pointer is the matrix destination. */
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
tb_data(tb, u4_(look_at)); /* Binds_ResolveLookAtPopAndTrans.look_at (MT3_S2S4*) */
}
}
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf)
{
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
// Pad Input
{
tb.used = 0; tb_scope_run(& tb) {
// Grab latest state from bios.
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]);
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]);
tb_emit_(pad_input_cam);
tb_data_(state, & smem.pad[0]);
tb_data_(cam, & smem.cam);
// tb_emit_(pad_input_cube_rotation);
// tb_data_(state, & smem.pad[0]);
// tb_data_(cube_rot, & smem.cube.rot);
// tb_data_(floor_rot, & smem.floor.rot);
}
}
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
// Update the position based on acceleration and velocity
gknown V3_S4_R pos = & smem.cube.pos;
gknown V3_S4_R vel = & smem.cube.vel;
gknown V3_S4_R acc = & smem.cube.accel;
add_v3s4(vel, acc[0]);
add_v3s4_fp(pos, vel[0]);
// vel->x += acc->x;
// vel->y += acc->y;
// vel->z += acc->z;
// pos->x += vel->x;
// pos->y += vel->y;
// pos->z += vel->z;
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
// Prep
S4 nclip = 0;
S4 orderingtbl_z = 0;
A2_S2 p; //???
S4 flag; //????
// Camera Look at (Tape) + inline C11 fallback — bundle runs, then C11 inlines the look_at.
// Currently: bundle's atom 0 (input_and_sub) runs + C11 does the rest. As bundle atoms
// are incrementally fixed, the corresponding C11 lines get commented out.
if (1)
{
tb.used = 0; tb_scope_run(& tb) {
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
}
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
// Preconditions: eye != target, up_in not collinear with (target - eye).
V3_S4 right, up, forward;
V3_S4 ux, uy, uz;
V3_S4 pos, off;
// forward = smem.cube.pos; sub_v3s4(& forward, smem.cam.pos); // RGA(Lengyel): Affine point - point = zero-weight direction. (now done by bundle atom 0)
// Read fwd from scratchpad[+0] (atom 0's output)
forward.x = u4_v(0x1F800000)[0];
forward.y = u4_v(0x1F800000)[1];
forward.z = u4_v(0x1F800000)[2];
forward.pad = u4_v(0x1F800000)[3];
// normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization. (now done by bundle atom 1)
// Read uz from scratchpad[+16] (atom 1's output)
uz.x = u4_v(0x1F800010)[0];
uz.y = u4_v(0x1F800010)[1];
uz.z = u4_v(0x1F800010)[2];
uz.pad = u4_v(0x1F800010)[3];
cross_v3s4(& uz, & v3s4(0, -fp_one, 0), & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
smem.cam.look_at.m[0][0] = ux.x; smem.cam.look_at.m[0][1] = ux.y; smem.cam.look_at.m[0][2] = ux.z;
smem.cam.look_at.m[1][0] = uy.x; smem.cam.look_at.m[1][1] = uy.y; smem.cam.look_at.m[1][2] = uy.z;
smem.cam.look_at.m[2][0] = uz.x; smem.cam.look_at.m[2][1] = uz.y; smem.cam.look_at.m[2][2] = uz.z;
pos = smem.cam.pos; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
mul_m3s2_v3s4(& smem.cam.look_at, & pos, & off);
trans_m3s2( & smem.cam.look_at, & off);
}
// Draw cube
if (1)
{
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
// Combine world and look_at matrix.
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view);
// gte_matrix_set_rotation (& smem.tform_world);
// gte_matrix_set_translation(& smem.tform_world);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
tb.used = 0; tb_scope(& tb) {
tb_emit(& tb, rbind_cube_g4_face);
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.cube.faces));
tb_data(& tb, u4_(smem.cube.verts));
tb_data(& tb, u4_(ordering_buf));
for (U4 i = 0; i < Cube_num_faces; i++) {
// Two triangles per quad face: (x,y,z) and (x,z,w)
tb_emit(& tb, cube_g4_face);
}
tb_emit(& tb, sync_primitive_arena);
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
// smem.cube.rot.y += 30;
}
// Draw floor
if (1)
{
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
// Combine world and look_at matrix.
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
// The tape atoms in-flight should not need to care.
// Prepare the tape. (Push protocol to tape)
tb.used = 0; tb_scope(& tb) {
// tb_emit(& tb, set_gte_mt3s2s4);
// tb_data(& tb, u4_(& smem.tform_view));
tb_emit(& tb, rbind_floor_f3_face);
// TODO(Ed): Just use a single context struct ref?
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.floor.faces));
tb_data(& tb, u4_(smem.floor.verts));
tb_data(& tb, u4_(ordering_buf));
for (U4 i = 0; i < Floor_num_faces; i++) {
tb_emit(& tb, floor_f3_face);
}
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
tb_emit(& tb, sync_primitive_arena);
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
// C-side state (pa->used) has already been updated by the tape!
// smem.floor.rot.y += 5;
}
}
GCC_OPTIMIZATION_ENABLE
void render(void) {
}
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
draw_sync(0);
vsync(0);
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
{
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
pa->used = 0;
}
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
}
GCC_OPTIMIZATION_DISABLE
int main(void)
{
smem = (SMemory){0};
smem.scratchpad = C_(U4_V, 0x1F800000);
// smem.primitives.used = 0;
// smem.active_buf_id = 0;
smem.cam.pos = v3s4(500, -1000, -1500);
/*Persistent Entity Setup*/{
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
Ent_Cube* cube = & smem.cube;
cube->rot = v3s2(0, 0, 0);
cube->scale = v3s4_fp_one();
cube->accel = v3s4(0, 1, 0);
cube->pos = v3s4(0, -400, 1800);
}
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
Ent_Floor* floor = & smem.floor;
floor->rot = v3s2(0, 0, 0);
floor->pos = v3s4(0, 450, 1800);
floor->scale = v3s4_fp_one();
}
}
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
reset_graph(0);
/* Direct BIOS: poll both ports during VBlank. */
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
resolve_look_at_init();
/* Pinned registers for the GPU init atom. */
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
tb.used = 0; tb_scope_run(& tb) {
tb_emit(& tb, screen_env_init);
tb_emit(& tb, gp_screen_init);
}
}
while (1) {
gknown S4* active_buf_id = & smem.active_buf_id;
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
gknown PrimitiveArena* pa = & smem.primitives;
update(pa, ordering_buf);
render();
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
};
return 0;
}
GCC_OPTIMIZATION_ENABLE
+102
View File
@@ -0,0 +1,102 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "duffle/dsl.h"
# include "duffle/math.h"
# include "duffle/gp.h"
# include "duffle/pad.h"
#endif
enum {
// PrimitiveBuff_Len = 4096,
// OrderingTbl_Len = 2048,
PrimitiveBuff_Len = 131072,
OrderingTbl_Len = 8192,
};
enum {
ScreenRes_X = 320,
ScreenRes_Y = 240,
ScreenZ = 320,
ScreenRes_CenterX = (ScreenRes_X >> 1),
ScreenRes_CenterY = (ScreenRes_Y >> 1),
};
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2);
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
typedef Array_(PrimitiveBuffer, 2);
typedef Struct_(PrimitiveArena) {
A2_PrimitiveBuffer buf;
U4 used;
};
#define Cube_num_verts 8
typedef Array_(V3_S2, Cube_num_verts);
#define Cube_num_faces 6
typedef Array_(V4_S2, Cube_num_faces);
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
{ -128, -128, -128 },
{ 128, -128, -128 },
{ 128, -128, 128 },
{ -128, -128, 128 },
{ -128, 128, -128 },
{ 128, 128, -128 },
{ 128, 128, 128 },
{ -128, 128, 128 }
};
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
{ 3, 2, 0, 1 },
{ 0, 1, 4, 5 },
{ 4, 5, 7, 6 },
{ 1, 2, 5, 6 },
{ 2, 3, 6, 7 },
{ 3, 0, 7, 4 },
};
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
return;
}
typedef Struct_(Ent_Cube) {
V3_S4 accel;
V3_S4 vel;
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S4 scale;
V3_S2 rot;
A8_V3_S2 verts;
A6_V4_S2 faces;
};
#define Floor_num_verts 4
typedef Array_(V3_S2, Floor_num_verts);
#define Floor_num_faces 2
typedef Array_(V3_S2, Floor_num_faces);
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
{ -900, 0, -900 },
{ -900, 0, 900 },
{ 900, 0, -900 },
{ 900, 0, 900 },
};
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
{ 0, 1, 2 },
{ 1, 3, 2 },
};
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
};
typedef Struct_(Ent_Floor) {
V3_S4 accel;
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S4 scale;
V3_S2 rot;
A4_V3_S2 verts;
A2_V3_S2 faces;
};
typedef Struct_(Camera) {
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S2 rot;
MT3_S2S4 look_at;
};
@@ -5,10 +5,10 @@
#pragma region hello_gte_tape #pragma region hello_gte_tape
// --- atom: cube_g4_face (87 words) --- // --- atom: cube_g4_face (77 words) ---
#define _atom_offset_cull_cube_g4_face_exit 48 #define _atom_offset_cull_cube_g4_face_exit 42
#define _atom_offset_bounds_chk_cube_g4_face_exit 12 #define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum { enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit, atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
@@ -18,7 +18,7 @@ enum {
// --- atom: floor_f3_face (58 words) --- // --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25 #define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 13 #define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum { enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit, atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
+29
View File
@@ -0,0 +1,29 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Source: C:\projects\Pikuma\ps1\code\hello_gte\hello_gte.tape.c
#pragma once
#pragma region hello_gte.tape
// --- atom: cube_g4_face (77 words) ---
#define _atom_offset_cull_cube_g4_face_exit 42
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
};
// --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
};
#pragma endregion hello_gte.tape
@@ -14,16 +14,16 @@
#include "duffle/gp.h" #include "duffle/gp.h"
#include "duffle/gte.h" #include "duffle/gte.h"
# include "duffle/gen/duffle.macs.h" # include "duffle/gen/macs.h"
# include "duffle/gen/duffle.offsets.h" # include "duffle/gen/offsets.h"
#include "duffle/atom_dsl.h" #include "duffle/atom_dsl.h"
#include "duffle/lottes_tape.h" #include "duffle/lottes_tape.h"
#include "duffle/word_count.metadata.h" #include "duffle/word_count.metadata.h"
# include "gen/gte_hello.offsets.h" # include "gen/offsets.h"
#include "hello_gte.h" #include "hello_gte.h"
#include "hello_gte_tape.c" #include "hello_gte.tape.c"
typedef U4 OrderingTable_Buffer[OrderingTbl_Len]; typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2); typedef Array_(OrderingTable_Buffer, 2);
@@ -122,7 +122,7 @@ global SMemory smem;
extern SMemory smem; extern SMemory smem;
// TODO(Ed): // TODO(Ed):
FI_ U4* spad_warm(MipsAtom atom) { FI_ U4* spad_warm(Slice_MipsCode atom) {
return nullptr; return nullptr;
} }
+218
View File
@@ -0,0 +1,218 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "duffle/gen/macs.h"
# include "duffle/gen/offsets.h"
# include "duffle/atom_dsl.h"
# include "duffle/lottes_tape.h"
# include "duffle/word_count.metadata.h"
# include "gen/offsets.h"
# include "hello_gte.h"
#endif
#pragma region MACs (Mips Atom components)
#pragma endregion MACs
#pragma region Baked Atoms
/* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() };
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0),
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
store_word( R_AT, R_T7, 4),
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
add_ui( R_T1, R_0, 10),
shift_lleft_self(R_T1, S_(U4)/2),
add_u_self( R_T1, R_T6),
load_word( R_AT, R_T1, 0),
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
store_word( R_AT, R_T7, 0),
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
or_u_self( R_AT, R_V0),
store_word( R_AT, R_T1, 0),
add_ui(R_T7, R_T7, 20),
mac_yield()
};
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) {
/* Load 3 indices */
load_half_u(R_T0, R_T4, 0),
load_half_u(R_T1, R_T4, 2),
load_half_u(R_T2, R_T4, 4),
/* Load Vertices into GTE */
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
/* Run Math */
nop2, gte_cmdw_rtpt,
nop2, gte_cmdw_nclip,
nop2,
/* Advance Face Cursor and Yield */
add_ui(R_T4, R_T4, 8),
mac_yield()
};
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(),
gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3(),
gte_cmdw_avg_sort_z4,
gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(),
mac_format_g4_color(
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
// end: branch(bounds_chk)
// end: branch(cull)
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
// atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices( R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < OrderingTbl_Len (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk)
// end: branch(culling)
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
#pragma endregion Baked Atoms
+41
View File
@@ -0,0 +1,41 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad/
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
#ifndef WORD_COUNT
#define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif
#define mac_put_disp_env(reg_transfer, reg_base, port) \
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
WORD_COUNT(mac_put_disp_env, 5)
#define mac_put_draw_env(reg_transfer, reg_base, port) \
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
WORD_COUNT(mac_put_draw_env, 16)
+76
View File
@@ -0,0 +1,76 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad\
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
#pragma once
#pragma region hello_joypad
// --- atom: cube_g4_face (76 words) ---
#define _atom_offset_cull_cube_g4_face_exit 41
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
};
// --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
};
// --- atom: pad_bios_snapshot (78 words) ---
#define _atom_offset_snap_root_skip_disconnected 8
#define _atom_offset_disconnected_snap_end 61
#define _atom_offset_case_2_id_dispatch 8
#define _atom_offset_pending_snap_end 51
#define _atom_offset_id_dispatch_try_analog_stick 11
#define _atom_offset_id_dispatch_snap_end 38
#define _atom_offset_try_analog_stick_try_analog_pad 12
#define _atom_offset_analog_stick_snap_end 24
#define _atom_offset_try_analog_pad_try_unsupported 11
#define _atom_offset_analog_pad_snap_end 10
enum {
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
};
// --- atom: pad_apply_input (60 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6
#define _atom_offset_dead_zone_low_check_dead_low_active 8
#define _atom_offset_dead_zone_high_check_dead_high_active 15
#define _atom_offset_dead_zone_skip_exit_stick 24
#define _atom_offset_end_low_exit_stick 12
enum {
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
};
#pragma endregion hello_joypad
+638
View File
@@ -0,0 +1,638 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "duffle/gen/macs.h"
# include "duffle/gen/offsets.h"
# include "duffle/dsl.atom.h"
# include "duffle/lottes_tape.h"
# include "duffle/mips.h"
# include "duffle/gte.h"
# include "duffle/gp.h"
# include "duffle/pad.h"
# include "duffle/word_count.metadata.h"
# include "duffle/psyq.h"
# include "duffle/math.atom.c"
# include "duffle/mips.atom.c"
# include "duffle/gte.atom.c"
# include "duffle/gp.atom.c"
# include "duffle/psyq.atom.c"
# include "gen/offsets.h"
# include "gen/macs.h"
# include "hello_joypad.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
/*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References:
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
*
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
* ./toolchain/psyq-4_7/lib/libgpu.a
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
*
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
* code[8..10] = padding (NOP) — 3 words to fill the packet
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
* code[13..14] = padding (NOP) — completes the 16-word packet
*/
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
})
#pragma endregion MACs
#pragma region Baked Atoms
enum {
R_ScreenX = R_T5 atom_reg atom_type(U2),
R_ScreenY = R_T6 atom_reg atom_type(U2),
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
#define R_ScreenBuf_Code R_T7_Code
};
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
) {
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
add_ui(R_T0, R_0, gp0_tpage_default),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
add_ui(R_T0, R_0, 1),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
add_ui(R_T0, R_0, 7),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
mac_yield(),
};
enum {
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
#define R_IO_BaseAddr_Code R_T4_Code
};
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU→GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
/* GP1: DisplayMode + Display Ranges */
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_yield(),
};
enum {
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
#define R_PrimCursor_Code R_T7_Code
#define R_FaceCursor_Code R_T4_Code
#define R_VertBase_Code R_T5_Code
#define R_OtBase_Code R_T6_Code
};
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later, only on the body path. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(R_PrimCursor),
gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3(R_PrimCursor),
gte_cmdw_avg_sort_z4,
gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
mac_format_g4_color(R_PrimCursor,
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
// end: branch(bounds_chk)
// end: branch(cull)
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
// atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(R_PrimCursor),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk)
// end: branch(culling)
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
/* ----- pad_bios_snapshot -----
* Per-frame snapshot of one BIOS pad buffer into PadState.
* Decoder (branch ladder on raw[0] status + raw[1] id):
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
* 6. else -> Unsupported (buttons=0, axes=0x80)
*
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
*
* Register use (atom-local; no wave-context touched):
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
* R_T1 = state base (kept throughout; all stores go through R_T1)
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
* R_T4 = scratch (shifts, compares, immediate loads, store values)
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
*/
enum {
R_PadRaw = R_T0 atom_reg atom_type(U1),
R_PadState = R_T1 atom_reg,
R_RawStatus = R_T2 atom_reg,
R_RawId = R_T3 atom_reg,
};
typedef Struct_(Binds_PadBiosSnapshot) {
PadBiosRaw* raw;
PadState* state;
};
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
) {
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
/* === Read raw[0] (status) + raw[1] (id) */
load_byte_u(R_RawStatus, R_PadRaw, 0),
load_byte_u(R_RawId, R_PadRaw, 1),
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
atom_label(disconnected) /* === Disconnected body. */
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(disconnected, snap_end)),
/* BD-slot: load next atom's entry point (replaces the nop).
* The unconditional branch always jumps to snap_end, where mac_yield_tail()
* transfers control to R_AtomJmp without re-loading it. */
mac_yield_load(),
atom_label(skip_disconnected)
/* === Case 2: Pending (status == 0 && id == 0)
* Combined check: if (status | id) != 0 then skip to id_dispatch.
* Falls through to the Pending case only when both are zero. */
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
atom_label(pending) /* === Pending body */
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(pending, snap_end)),
mac_yield_load(),
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
store_word( R_T5, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0x41),
store_byte( R_T4, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(id_dispatch, snap_end)),
mac_yield_load(),
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
atom_label(analog_stick) /* === AnalogStick body
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
store_half( R_T4, R_PadState, O_(PadState,right_x)),
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
store_byte( R_T5, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(analog_stick, snap_end)),
mac_yield_load(),
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
and_i( R_T4, R_RawId, 0xF0),
add_ui( R_T5, R_0, 0x70),
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
atom_label(analog_pad) /* === AnalogPad body
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
store_half( R_T4, R_PadState, O_(PadState,right_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(analog_pad, snap_end)),
mac_yield_load(),
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
add_ui( R_T4, R_0, PadStatus_Unsupported),
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
store_byte( R_T4, R_PadState, O_(PadState,id)),
/* Fall through to snap_end. */
atom_label(no_jump_fallthrough)
mac_yield_load(),
atom_label(snap_end)
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
mac_yield_tail(),
};
/* ----- pad_apply_input -----
* Reads pad[0].buttons + pad[0].left_x;
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
* - Analog stick X (dead zone 0x70..0x90):
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
* - D-pad + analog deltas add when used together.
*
* Convention:
* pad_state = 0 means no buttons active.
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
*
* Signed-delta trick:
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
*/
typedef Struct_(Binds_PadApplyInput) {
PadState* state;
V3_S2* cube_rot;
V3_S2* floor_rot;
};
enum {
R_PadStateT5 = R_T5 atom_reg,
R_CubeRot = R_T1 atom_reg,
R_FloorRot = R_T2 atom_reg,
};
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
, atom_writes( R_CubeRot, R_FloorRot)
) {
/* Pop Binds from tape (state, cube_rot, floor_rot) */
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
/* Load pad[0].buttons into R_T0. */
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
// Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30),
add_si( R_T3, R_T3, 5),
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30),
add_si( R_T3, R_T3, -5),
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_dpad_right)
/* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
add_ui( R_T4, R_0, 0x90),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
mac_yield_load(),
atom_label(dead_low_active)
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
/* delta = 0x80 - left_x (positive). */
/* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
jump_rel(atom_offset(end_low, exit_stick)),
mac_yield_load(),
atom_label(dead_high_active)
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
sub_u( R_T3, R_T4, R_T3),
/* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
atom_label(no_jump_fallthrough)
mac_yield_load(),
atom_label(exit_stick)
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
mac_yield_tail(),
};
#pragma endregion Baked Atoms
+441
View File
@@ -0,0 +1,441 @@
#pragma region Vendors
#include <stdio.h>
#include <stdlib.h>
#include <assert.h>
// #include "libgpu.h"
// #include "libetc.h"
// #include "libgte.h"
#pragma endregion Vendors
#pragma region Duffle Headers
# include "duffle/gen/macs.h"
# include "duffle/gen/offsets.h"
#include "duffle/word_count.metadata.h"
#include "duffle/dsl.h"
#include "duffle/memory.h"
#include "duffle/math.h"
#include "duffle/gcc_asm.h"
#include "duffle/mips.h"
#include "duffle/gp.h"
#include "duffle/gte.h"
#include "duffle/pad.h"
#include "duffle/dsl.atom.h"
#include "duffle/lottes_tape.h"
#include "duffle/psyq.h"
#pragma endregion Duffle Headers
#pragma region Duffle TUs
#include "duffle/math.atom.c"
#include "duffle/mips.atom.c"
#include "duffle/gte.atom.c"
#include "duffle/gp.atom.c"
#include "duffle/psyq.atom.c"
#pragma endregion Duffle TUs
#pragma region Joypade Headers
# include "gen/macs.h"
# include "gen/offsets.h"
#include "hello_joypad.h"
#pragma region Joypad Headers
#pragma region Hello Joypad TUs
#include "hello_joypad.atom.c"
#pragma endregion Hello Joypad TUs
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
};
typedef Struct_(SMemory) {
PrimitiveArena primitives;
A2_OrderingTable_Buffer ordering_tbl;
DoubleBuffer screen_buf;
S4 active_buf_id;
U4 MemTape[MemTape_Len];
M3_S2 tform_world;
Ent_Cube cube;
Ent_Floor floor;
PadBiosRaw pad_raw[2];
PadState pad[2];
U4_V scratchpad; // d-cache
};
global SMemory smem;
extern SMemory smem;
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
gknown PrimitiveArena* pa = & smem.primitives;
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
assert(pa->used + type_width < PrimitiveBuff_Len);
B1* next = buf + pa->used;
pa->used += type_width;
return next;
}
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
*
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
* The C-level writes after the call re-load the pointers from their callee-saved homes.
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
// Use enums.
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
* $a0 = raw0 (rgcc-bound; survives the sequence below)
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
u1_v(raw0)[0] = 0xFF;
u1_v(raw1)[0] = 0xFF;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
}
GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf)
{
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
if (0) // Pad Input (dead — kept for the source-as-written record; references the deleted `pad_state` field)
{
(void)Pad_Left; (void)Pad_Right; /* suppress unused-token warnings */
if (false) {
smem.cube.rot.y += 30;
smem.floor.rot.y += 5;
}
if (false) {
smem.cube.rot.y -= 30;
smem.floor.rot.y -= 5;
}
}
if (1) // Pad Input (Tape version)
{
tb.used = 0; tb_scope_run(& tb) {
/* BIOS-owned polling: per-frame snapshot of both ports. */
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]);
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]);
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
tb_emit_(pad_apply_input);
tb_data_(state, & smem.pad[0]);
tb_data_(cube_rot, & smem.cube.rot);
tb_data_(floor_rot, & smem.floor.rot);
}
}
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
// Update the position based on acceleration and velocity
gknown V3_S4_R pos = & smem.cube.pos;
gknown V3_S4_R vel = & smem.cube.vel;
gknown V3_S4_R acc = & smem.cube.accel;
add_v3s4(vel, acc[0]);
add_v3s4_fp(pos, vel[0]);
// vel->x += acc->x;
// vel->y += acc->y;
// vel->z += acc->z;
// pos->x += vel->x;
// pos->y += vel->y;
// pos->z += vel->z;
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
// Prep
S4 nclip = 0;
S4 orderingtbl_z = 0;
A2_S2 p; //???
S4 flag; //????
// Draw Cube
if (0)
{
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.cube.pos);
m3s2_scale (& smem.tform_world, & smem.cube.scale);
// gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
{
Poly_G4* quad = prim_alloc(Poly_G4); set_poly_g4(quad);
quad->c0 = rgb8(255, 0, 255);
quad->c1 = rgb8(255, 255, 0);
quad->c2 = rgb8( 0, 255, 255);
quad->c3 = rgb8( 0, 255, 0);
V4_S2* face = & smem.cube.faces[face_id];
V3_S2* p0 = & smem.cube.verts[face->x];
V3_S2* p1 = & smem.cube.verts[face->y];
V3_S2* p2 = & smem.cube.verts[face->z];
V3_S2* p3 = & smem.cube.verts[face->w];
nclip = rtp_avg_nclip_a4_v3s2(
p0, p1, p2, p3,
& quad->p0, & quad->p1, & quad->p2, & quad->p3,
& p, & orderingtbl_z, & flag
);
if (nclip <= 0) {
continue;
}
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], quad);
}
}
// smem.cube.rot.x += 6;
// smem.cube.rot.y += 8;
// smem.cube.rot.z += 12;
smem.cube.rot.y += 30;
}
// Draw cube (tape method) - two triangles per face
if (1)
{
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.cube.pos);
m3s2_scale (& smem.tform_world, & smem.cube.scale);
gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
tb.used = 0; tb_scope(& tb) {
tb_emit(& tb, rbind_cube_g4_face);
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.cube.faces));
tb_data(& tb, u4_(smem.cube.verts));
tb_data(& tb, u4_(ordering_buf));
for (U4 i = 0; i < Cube_num_faces; i++) {
// Two triangles per quad face: (x,y,z) and (x,z,w)
tb_emit(& tb, cube_g4_face);
}
tb_emit(& tb, sync_primitive_arena);
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run(tb_slice(tb));
// smem.cube.rot.y += 30;
}
// Draw Floor
if (0)
{
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.floor.pos);
m3s2_scale (& smem.tform_world, & smem.floor.scale);
gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
for (U4 face_id = 0; face_id < Floor_num_faces; face_id += 1)
{
Poly_F3* tri = prim_alloc(Poly_F3); set_poly_f3(tri);
tri->color = rgb8(255, 255, 255);
V3_S2* face = & smem.floor.faces[face_id];
register V3_S2* p0 rgcc(R_T4) = & smem.floor.verts[face->x];
register V3_S2* p1 rgcc(R_T5) = & smem.floor.verts[face->y];
register V3_S2* p2 rgcc(R_T6) = & smem.floor.verts[face->z];
gte_load_v0(p0, R_T4);
/*
asm volatile( ".word " "%0" ", %1" : :
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_xy & REG_MASK) << RT_SHIFT) | (0 & IMM_MASK)),
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_z & REG_MASK) << RT_SHIFT) | (GTE_Z_Offset & IMM_MASK)),
"r"(p0) :
"$2", "$8", "$9", "$31", "memory"
);
*/
gte_load_v1(p1, R_T5);
gte_load_v2(p2, R_T6);
gte_rtpt();
gte_nclip();
gte_stotz(& nclip);
// nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
// , & tri->p0, & tri->p1, & tri->p2
// , & p, & orderingtbl_z, & flag
// );
// if (nclip <= 0) {
// continue;
// }
if (nclip > 0 ) {
gte_stsxy3(& tri->p0, & tri->p1, & tri->p2);
gte_avsz3();
gte_stotz(& orderingtbl_z);
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
}
}
}
smem.floor.rot.y += 5;
}
// Draw floor tape method
if (1)
{
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.floor.pos);
m3s2_scale (& smem.tform_world, & smem.floor.scale);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
// The tape atoms in-flight should not need to care.
// Prepare the tape. (Push protocol to tape)
tb.used = 0; tb_scope(& tb) {
tb_emit(& tb, set_gte_world);
tb_data(& tb, u4_(& smem.tform_world));
tb_emit(& tb, rbind_floor_f3_face);
// TODO(Ed): Just use a single context struct ref
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.floor.faces));
tb_data(& tb, u4_(smem.floor.verts));
tb_data(& tb, u4_(ordering_buf));
for (U4 i = 0; i < Floor_num_faces; i++) {
tb_emit(& tb, floor_f3_face);
}
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
tb_emit(& tb, sync_primitive_arena);
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run(tb_slice(tb));// Fire off the tape.
// C-side state (pa->used) has already been updated by the tape!
// smem.floor.rot.y += 5;
}
}
GCC_OPTIMIZATION_ENABLE
void render(void) {
}
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
draw_sync(0);
vsync(0);
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
{
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
pa->used = 0;
}
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
}
GCC_OPTIMIZATION_DISABLE
int main(void)
{
smem = (SMemory){0};
smem.scratchpad = C_(U4_V, 0x1F800000);
// smem.primitives.used = 0;
// smem.active_buf_id = 0;
/*Persistent Entity Setup*/{
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
Ent_Cube* cube = & smem.cube;
cube->rot = v3s2(0, 0, 0);
cube->scale = v3s4_fp_one();
cube->accel = v3s4(0, 1, 0);
cube->pos = v3s4(0, -400, 1800);
}
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
Ent_Floor* floor = & smem.floor;
floor->rot = v3s2(0, 0, 0);
floor->pos = v3s4(0, 450, 1800);
floor->scale = v3s4_fp_one();
}
}
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
reset_graph(0);
/* Direct BIOS: poll both ports during VBlank. */
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
/* Pinned registers for the GPU init atom. */
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
tb.used = 0; tb_scope_run(& tb) {
tb_emit(& tb, screen_env_init);
tb_emit(& tb, gp_screen_init);
}
}
while (1) {
gknown S4* active_buf_id = & smem.active_buf_id;
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
gknown PrimitiveArena* pa = & smem.primitives;
update(pa, ordering_buf);
render();
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
};
return 0;
}
GCC_OPTIMIZATION_ENABLE
+102
View File
@@ -0,0 +1,102 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "duffle/dsl.h"
# include "duffle/math.h"
# include "duffle/gp.h"
# include "duffle/pad.h"
#endif
enum {
// PrimitiveBuff_Len = 4096,
// OrderingTbl_Len = 2048,
PrimitiveBuff_Len = 131072,
OrderingTbl_Len = 8192,
};
enum {
ScreenRes_X = 320,
ScreenRes_Y = 240,
ScreenZ = 320,
ScreenRes_CenterX = (ScreenRes_X >> 1),
ScreenRes_CenterY = (ScreenRes_Y >> 1),
};
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2);
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
typedef Array_(PrimitiveBuffer, 2);
typedef Struct_(PrimitiveArena) {
A2_PrimitiveBuffer buf;
U4 used;
};
#define Cube_num_verts 8
typedef Array_(V3_S2, Cube_num_verts);
#define Cube_num_faces 6
typedef Array_(V4_S2, Cube_num_faces);
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
{ -128, -128, -128 },
{ 128, -128, -128 },
{ 128, -128, 128 },
{ -128, -128, 128 },
{ -128, 128, -128 },
{ 128, 128, -128 },
{ 128, 128, 128 },
{ -128, 128, 128 }
};
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
{ 3, 2, 0, 1 },
{ 0, 1, 4, 5 },
{ 4, 5, 7, 6 },
{ 1, 2, 5, 6 },
{ 2, 3, 6, 7 },
{ 3, 0, 7, 4 },
};
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
return;
}
typedef Struct_(Ent_Cube) {
V3_S4 accel;
V3_S4 vel;
V3_S4 pos;
V3_S4 scale;
V3_S2 rot;
A8_V3_S2 verts;
A6_V4_S2 faces;
};
#define Floor_num_verts 4
typedef Array_(V3_S2, Floor_num_verts);
#define Floor_num_faces 2
typedef Array_(V3_S2, Floor_num_faces);
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
{ -900, 0, -900 },
{ -900, 0, 900 },
{ 900, 0, -900 },
{ 900, 0, 900 },
};
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
{ 0, 1, 2 },
{ 1, 3, 2 },
};
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
};
typedef Struct_(Ent_Floor) {
V3_S4 accel;
V3_S4 pos;
V3_S4 scale;
V3_S2 rot;
A4_V3_S2 verts;
A2_V3_S2 faces;
};
+659
View File
@@ -0,0 +1,659 @@
#if 0 /* ac_pad_sio_write_pad_state — superseded by pad_bios_snapshot */
/* ============================================================
* raw_sio_pad_poll_20260802 — superseded by bios_pad_buffer_snapshot_20260803.
* The doomed raw-SIO production atoms (ac_pad_sio_write_pad_state,
* pad_sio_init, pad_sio_step, pad_sio_diag_pin, pad_sio_diag_byte_exchange)
* reference symbols that were removed from code/duffle/pad.h during
* Phase 1. Each is wrapped in a narrow `#if 0` so the C compile skips
* the body while the source-as-written text stays in place for the
* Phase 5.1 deletion pass. The wrap is removed (and the bodies are
* deleted) by Phase 5.1 of this track.
* ============================================================ */
* Writes the per-port PadState in 5 instructions plus 4 store_word calls (status,
* buttons, left_x/y/right_x/right_y packed, attempt). The provisional decode publishes
* 0x0000FFFF buttons + centered axes on every path until response-byte decode lands.
*
* Args:
* status_val - the PadSioStatus enum value to publish
* state_ptr_reg - the PadState* base (R_PadState at the call site)
* scratch_reg - scratch register for the value being stored (e.g., R_T0)
*
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
* two-instruction zero-extended buttons load).
*/
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
add_ui(scratch_reg, R_0, status_val),
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
* libetc convention. Build it with LUI + ORI so addiu does not
* sign-extend 0xFFFF to 0xFFFFFFFF. */
load_upper_i(scratch_reg, 0x0000),
or_i(scratch_reg, scratch_reg, 0xFFFF),
store_word(scratch_reg, state_ptr_reg, O_(PadState,buttons)),
add_ui(scratch_reg, R_0, 0x80808080),
store_word(scratch_reg, state_ptr_reg, O_(PadState,left_x)),
add_ui(scratch_reg, R_0, 0),
store_word(scratch_reg, state_ptr_reg, O_(PadState,attempt))
})
#endif /* end ac_pad_sio_write_pad_state wrap */
/* ----- pad_sio_init -----
* Boot-time SIO0 init. Caller pins R_T6 = sio_base_addr0.
* Issues SIO CTRL=0x0040 (reset), MODE=0x000D, BAUD=0x0088.
* (Phase 2 fills the body.)
*/
#if 0 /* pad_sio_init — superseded by pad_bios_init_start (Phase 1.3) */
internal MipsAtom_(pad_sio_init) atom_info(atom_phase(pad_init)
, atom_reads(R_T5, R_T6)
, atom_writes(R_T5, R_T6)
) {
/* FIX 2026-08-02: explicitly load the KSEG1 base into R_T6 at the top of
* the atom body. The rgcc(R_PadSioBase) binding in main() pins R_T6 = base
* when main() runs, but $12 is caller-saved per the O32 ABI — when tape_run
* is invoked, R_T6 is fair game. The atom body cannot rely on the value. */
load_upper_i(R_T6, pad_IO_KSEG1_BASE >> 16), /* R_T6 high 16 = 0xBF80 */
or_i(R_T6, R_T6, pad_IO_KSEG1_BASE & 0xFFFF), /* R_T6 = 0xBF800000 */
/* SIO CTRL = 0x0040 (reset) */
add_ui(R_T5, R_0, pad_SIO_CTRL_RESET),
store_half(R_T5, R_T6, pad_SIO_CTRL_OFFSET),
/* SIO MODE = 0x000D (MUL1, 8-bit, no parity, idle-high) */
add_ui(R_T5, R_0, pad_SIO_MODE_INIT),
store_half(R_T5, R_T6, pad_SIO_MODE_OFFSET),
/* SIO BAUD = 0x0088 (~250 kHz) */
add_ui(R_T5, R_0, pad_SIO_BAUD_INIT),
store_half(R_T5, R_T6, pad_SIO_BAUD_OFFSET),
mac_yield(),
};
#endif /* end pad_sio_init wrap */
/* ----- pad_sio_step -----
* Per-frame bounded raw-SIO transaction. Reads PadState pointers + SIO
* base addresses from Binds_PadSioStep; writes per-port status +
* buttons + axes into smem.pad[0..1].
* Body shape (per spec §"Transaction model (per port, per pad_sio_step)"):
* port 0: CTRL=CLEANUP → settle → CTRL=port-select → settle → exchange 5
* bytes (addr + 0x42 0x00 0x00 0x00) → decode → write PadState[0]
* → CTRL=CLEANUP.
* port 1: swap scratch regs (sio_base_addr1 → R_PadSioBase, state1 →
* R_PadState) → mirror port 0 sequence.
*
* Bounded-loop semantics: every countdown is wrapped in
* add_ui_self(R_T1, -1) + branch_ne(R_T1, R_0, ...)
* with a known maximum (pad_SIO_SETTLE_BEFORE_TX=1000, pad_SIO_SETTLE_AFTER_TX=2000,
* pad_SIO_WAIT_BUDGET=4096). The static-analysis pass currently reports
* has_loops = true; the follow-up metaprogram track that learns modeled-bounded
* loops is out of scope here (per spec §"Risks").
*
* Scratch register strategy:
* R_PadStatus = R_T4 — RESERVED for port-1 swap (holds state1)
* R_PadCountdown = R_T5 — RESERVED for port-1 swap (holds sio_base_addr1)
* R_T0 — byte-exchange value + STAT read (clobbered freely)
* R_T1 — countdown budget (clobbered freely)
* R_PadState = R_T7 — PadState* (preserved for PadState writes)
* R_PadSioBase = R_T6 — SIO base (preserved through the port)
*
* Response decode (Task 3.1 teaching scope):
* - status = PadSioStatus_Digital (hardcoded)
* - buttons = 0xFFFF (no buttons pressed in the provisional libetc
* convention; full response-byte decode is follow-up)
* - axes = 0x80808080 (centered: left_x=0x80, left_y=0x80,
* right_x=0x80, right_y=0x80)
* - attempt = 0
* - DualShock handshake (0x43 0x01 → 0x44 0x01 0x03 → 0x43 0x00) is
* follow-up scope; the hardcoded digital decode is a placeholder.
*
* Both ports raise /CS (CTRL = pad_SIO_CTRL_CLEANUP) before exit. Both ports
* treat response timeout as PadSioStatus_Disconnected per the spec §"Failure
* handling" + the canonical per-port timeout semantics.
*/
#if 0 /* pad_sio_step — superseded by pad_bios_snapshot (Phase 2.1) */
internal MipsAtom_(pad_sio_step) atom_info(atom_bind(Binds_PadSioStep)
, atom_reads(R_TapePtr, R_PadSioBase, R_PadState, R_PadStatus, R_PadCountdown)
, atom_writes(R_PadStatus, R_PadCountdown)
) {
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
* top. The rgcc() binding in main() does NOT survive the tape_run call
* because R_T6 is caller-saved per the O32 ABI. The pad_sio_init atom
* (also in the per-frame tape) reloads R_T6 separately. */
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
/* Pop Binds from tape (in Binds_PadSioStep declaration order) */
load_word(R_PadState, R_TapePtr, O_(Binds_PadSioStep,state0)),
load_word(R_PadStatus, R_TapePtr, O_(Binds_PadSioStep,state1)), /* reserved for port-1 swap */
load_word(R_PadSioBase, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr0)),
load_word(R_PadCountdown, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr1)), /* reserved for port-1 swap */
add_ui_self(R_TapePtr, S_(Binds_PadSioStep)),
/* ============== PORT 0 TRANSACTION ============== */
/* Use R_T0 (byte value / STAT read) + R_T1 (countdown) as scratch.
* R_PadStatus (state1) + R_PadCountdown (sio_base_addr1) are preserved
* through the port-0 body and swapped into R_PadSioBase + R_PadState
* at atom_offset(port1_start, ...) below. */
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
atom_label(settle_pre_port0)
nop, /* BD slot */
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_pre_port0, settle_pre_port0)),
/* 2. Port-select: CTRL = 0x0003 (TX enable + DTR /CS) for port 0 */
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS), /* set /CS line low */
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
atom_label(settle_post_port0)
nop,
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_post_port0, settle_post_port0)),
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
add_ui(R_T0, R_0, pad_PROTO_ADDR),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack0_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port0, ack0_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack0_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port0, wait_ack0_port0)),
/* RX timeout → mark disconnected; skip to port 1 */
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack0)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack0, port1_start)),
atom_label(ack0_received_port0)
/* Read open-bus response byte 0 — discard per docs/psx-spx §controllersandmemorycards.md */
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Confirm RX FIFO drained before sending byte 1. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel0_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port0, ack_released_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel0_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port0, wait_ackrel0_port0)),
/* RX-drain timeout → disconnected; skip to port 1 */
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel0)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel0, port1_start)),
atom_label(ack_released_port0)
/* === Byte 1 (port 0): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack1_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port0, ack1_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack1_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port0, wait_ack1_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack1)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack1, port1_start)),
atom_label(ack1_received_port0)
/* Read response ID byte — discarded for teaching scope (decode hardcoded). */
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* RX FIFO drain wait. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel1_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port0, ack_released1_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel1_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port0, wait_ackrel1_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel1)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel1, port1_start)),
atom_label(ack_released1_port0)
/* === Byte 2 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack2_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port0, ack2_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack2_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port0, wait_ack2_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack2)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack2, port1_start)),
atom_label(ack2_received_port0)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel2_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port0, ack_released2_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel2_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port0, wait_ackrel2_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel2)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel2, port1_start)),
atom_label(ack_released2_port0)
/* === Byte 3 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack3_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port0, ack3_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack3_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port0, wait_ack3_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack3)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack3, port1_start)),
atom_label(ack3_received_port0)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel3_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port0, ack_released3_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel3_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port0, wait_ackrel3_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel3)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel3, port1_start)),
atom_label(ack_released3_port0)
/* === Byte 4 (FINAL, port 0): send 0x00 + RX-not-empty wait + read final byte === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_rx4_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port0, rx4_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_rx4_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port0, wait_rx4_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_rx4)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_rx4, port1_start)),
atom_label(rx4_received_port0)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
/* === RESPONSE DECODE (hardcoded for teaching scope) ===
* Per the plan §"Phase 3 task 3.1" + spec §"Architecture":
* - Full decode (buttons/axes from response bytes) is follow-up scope.
* - Teaching scope: hardcode digital poll response.
* status = PadSioStatus_Digital
* buttons = 0x0000FFFF (no buttons pressed — placeholder)
* axes = 0x80808080 (left_x=0x80, left_y=0x80, right_x=0x80, right_y=0x80)
* attempt = 0
*/
atom_label(decode_port0)
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
/* /CS cleanup: raise /CS, clear stale status before exiting port 0. */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* ============== PORT 1 SETUP ============== */
/* Swap: R_PadCountdown holds sio_base_addr1; R_PadStatus holds state1. */
atom_label(port1_start)
add_u(R_PadSioBase, R_0, R_PadCountdown), /* sio_base_addr1 → R_PadSioBase */
add_u(R_PadState, R_0, R_PadStatus), /* state1 → R_PadState */
/* ============== PORT 1 TRANSACTION (mirror of port 0) ============== */
/* R_PadStatus + R_PadCountdown are no longer reserved (port 1 is the
* last transaction); we still use R_T0/R_T1 as scratch to match port 0. */
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
atom_label(settle_pre_port1)
nop,
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_pre_port1, settle_pre_port1)),
/* 2. Port-select: CTRL = 0x0003 | (1 << 13) (port 1 select) */
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
or_i(R_T0, R_T0, 1 << 13), /* port 1 select bit (CTRL bit 13 = port select) */
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
atom_label(settle_post_port1)
nop,
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_post_port1, settle_post_port1)),
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
add_ui(R_T0, R_0, pad_PROTO_ADDR),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack0_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port1, ack0_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack0_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port1, wait_ack0_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack0)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack0, end_atom)),
atom_label(ack0_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel0_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port1, ack_released_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel0_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port1, wait_ackrel0_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel0)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel0, end_atom)),
atom_label(ack_released_port1)
/* === Byte 1 (port 1): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack1_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port1, ack1_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack1_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port1, wait_ack1_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack1)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack1, end_atom)),
atom_label(ack1_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel1_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port1, ack_released1_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel1_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port1, wait_ackrel1_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel1)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel1, end_atom)),
atom_label(ack_released1_port1)
/* === Byte 2 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack2_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port1, ack2_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack2_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port1, wait_ack2_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack2)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack2, end_atom)),
atom_label(ack2_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel2_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port1, ack_released2_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel2_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port1, wait_ackrel2_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel2)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel2, end_atom)),
atom_label(ack_released2_port1)
/* === Byte 3 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack3_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port1, ack3_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack3_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port1, wait_ack3_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack3)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack3, end_atom)),
atom_label(ack3_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel3_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port1, ack_released3_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel3_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port1, wait_ackrel3_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel3)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel3, end_atom)),
atom_label(ack_released3_port1)
/* === Byte 4 (FINAL, port 1): send 0x00 + RX-not-empty wait + read final byte === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_rx4_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port1, rx4_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_rx4_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port1, wait_rx4_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_rx4)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_rx4, end_atom)),
atom_label(rx4_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
/* === RESPONSE DECODE (port 1) === */
atom_label(decode_port1)
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
/* /CS cleanup: raise /CS, clear stale status before exiting port 1. */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
atom_label(end_atom)
mac_yield(),
};
#endif /* end pad_sio_step wrap */
/* ----- pad_sio_diag_pin -----
* Per-frame diagnostic counter. The caller binds R_DiagPinScratch to
* scratch_for_atom_diag_pin for temporary gdb verification.
*/
#if 0 /* pad_sio_diag_pin — superseded (raw-SIO phase removed) */
internal MipsAtom_(pad_sio_diag_pin) atom_info(atom_phase(pad_init)
, atom_reads(R_T0, R_T1, R_DiagPinScratch)
, atom_writes(R_T0, R_T1, R_DiagPinScratch)
) {
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
load_upper_i(R_DiagPinScratch, 0x8001),
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
/* High half = 0xD1A6; low half increments once per atom invocation. */
load_word(R_T1, R_DiagPinScratch, 0),
nop,
add_ui(R_T1, R_T1, 1),
and_i(R_T0, R_T1, 0xFFFF),
load_upper_i(R_T1, 0xD1A6),
or_i(R_T1, R_T1, 0),
or_u(R_T1, R_T1, R_T0),
store_word(R_T1, R_DiagPinScratch, 0),
mac_yield(),
};
#endif /* end pad_sio_diag_pin wrap */
/* ----- pad_sio_diag_byte_exchange -----
* Temporary two-byte wire probe: sends 0x01 and 0x42, then stores the
* open-bus byte and response ID in scratch_for_atom_diag_pin.
*/
#if 0 /* pad_sio_diag_byte_exchange — superseded (raw-SIO phase removed) */
internal MipsAtom_(pad_sio_diag_byte_exchange) atom_info(atom_phase(pad_init)
, atom_reads(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
, atom_writes(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
) {
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
load_upper_i(R_DiagPinScratch, 0x8001),
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
* top. The rgcc() binding in main() does NOT survive the tape_run call
* because R_T6 is caller-saved per the O32 ABI. */
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
add_ui(R_T0, R_0, pad_PROTO_ADDR),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(diag_wait_ack0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack0, diag_ack0_done)),
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack0, diag_wait_ack0)),
add_ui(R_T0, R_0, 0xDEADAC01),
store_word(R_T0, R_DiagPinScratch, 0),
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack0, diag_timeout)),
atom_label(diag_ack0_done)
load_byte_u(R_T2, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(diag_wait_ack1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack1, diag_ack1_done)),
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack1, diag_wait_ack1)),
add_ui(R_T0, R_0, 0xDEADAC02),
store_word(R_T0, R_DiagPinScratch, 0),
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack1, diag_timeout)),
atom_label(diag_ack1_done)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
nop,
shift_lleft(R_T0, R_T0, 8),
or_u(R_T2, R_T2, R_T0),
store_word(R_T2, R_DiagPinScratch, 0),
atom_label(diag_success)
branch_equal(R_0, R_0, atom_offset(diag_success, diag_done)),
nop,
atom_label(diag_timeout_ack0)
add_ui(R_T0, R_0, 0xDEADAC01),
store_word(R_T0, R_DiagPinScratch, 0),
atom_label(diag_timeout_ack1)
add_ui(R_T0, R_0, 0xDEADAC02),
store_word(R_T0, R_DiagPinScratch, 0),
atom_label(diag_timeout)
add_ui(R_T0, R_0, 0xDEADACFF),
store_word(R_T0, R_DiagPinScratch, 0),
atom_label(diag_done)
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
mac_yield(),
};
#endif /* end pad_sio_diag_byte_exchange wrap */
+7 -1
View File
@@ -7,12 +7,12 @@ A rest from the usual.
## Dependencies ## Dependencies
I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...): I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...):
![system_info](./docs/assets/system_info.png)
[armips](https://github.com/Kingcom/armips) [armips](https://github.com/Kingcom/armips)
* Supports doing bare-metal assembly for the ps1 * Supports doing bare-metal assembly for the ps1
* `scoop install armips` or just clone and build.. * `scoop install armips` or just clone and build..
* Was used early in the course. Now I just use an macro asm dsl in C11.
[luajit-2.1](https://github.com/LuaJIT/LuaJIT.git) [luajit-2.1](https://github.com/LuaJIT/LuaJIT.git)
@@ -73,3 +73,9 @@ scoop install luajit
![hello_psyq!](./docs/assets/pcsx-redux_2025-08-05_23-01-19.png) ![hello_psyq!](./docs/assets/pcsx-redux_2025-08-05_23-01-19.png)
![cube!](./docs/assets/pcsx-redux_2025-10-11_03-04-01.png) ![cube!](./docs/assets/pcsx-redux_2025-10-11_03-04-01.png)
![cube and floor!](./docs/assets/pcsx-redux_2026-07-10_22-47-02.png) ![cube and floor!](./docs/assets/pcsx-redux_2026-07-10_22-47-02.png)
Win 11 machine:
![system_info](./docs/assets/system_info.png)
Still haven't gotten around to trying this on linux...
+176 -79
View File
@@ -180,29 +180,15 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map) $link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
$link_args += ($f_link_pass_through_prefix + $f_link_start_group) $link_args += ($f_link_pass_through_prefix + $f_link_start_group)
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
$libraries = @( $libraries = @(
"api", "api",
"c", "c",
"c2",
"card",
"cd",
"comb",
"ds",
"etc", "etc",
"gpu", "gpu",
"gs", "gte"
"gte",
"gun",
"hmd",
"math",
"mcrd",
"mcx",
"pad",
"press",
"sio",
"snd",
"spu",
"tap"
) )
foreach ($lib in $libraries) { foreach ($lib in $libraries) {
$link_args += ($f_link_lib + $lib) $link_args += ($f_link_lib + $lib)
@@ -238,9 +224,7 @@ function ps1-meta { param(
[string[]]$passes = @('--pre-link'), [string[]]$passes = @('--pre-link'),
[string[]]$extra_args = @() [string[]]$extra_args = @()
) )
# `--unity-root` and `--source` are # `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
# be supplied; the other must be absent.
if ($null -ne $unity_root -and $unity_root -ne '') if ($null -ne $unity_root -and $unity_root -ne '')
{ {
if ($null -ne $sources -and $sources.Count -gt 0) { if ($null -ne $sources -and $sources.Count -gt 0) {
@@ -276,6 +260,73 @@ function ps1-meta { param(
} }
} }
function inject-dwarf { param(
[string]$elf,
[string]$path_gen
)
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
if (-not (Test-Path $path_dwarf_line_bin)) { return }
if (-not (Test-Path $path_dwarf_aranges_bin)) { return }
if (-not (Test-Path $path_dwarf_rnglists_bin)) { return }
Write-Host "[build] DWARF-injecting $elf -> $path_inject_elf"
Copy-Item -LiteralPath $elf -Destination $path_inject_elf -Force
# Objcopy call 1: 3x --update-section for the PC-mapping tables (line, aranges, rnglists).
$objcopy_args_dwarf_pc = @(
"--update-section=.debug_line=$path_dwarf_line_bin",
"--update-section=.debug_aranges=$path_dwarf_aranges_bin",
"--update-section=.debug_rnglists=$path_dwarf_rnglists_bin"
)
& $Objcopy @objcopy_args_dwarf_pc $path_inject_elf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] objcopy dwarf-pc splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
return
}
# Objcopy call 2: 3x --update-section + 2x --add-section for the debug-data tables (info, abbrev, str, loc, loclists).
$objcopy_args_dwarf_info = @(
"--update-section=.debug_info=$path_dwarf_info_bin",
"--update-section=.debug_abbrev=$path_dwarf_abbrev_bin",
"--update-section=.debug_str=$path_dwarf_str_bin",
"--add-section=.debug_loc=$path_dwarf_loc_bin",
"--add-section=.debug_loclists=$path_dwarf_loclists_bin"
)
& $Objcopy @objcopy_args_dwarf_info $path_inject_elf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] objcopy dwarf-info splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
return
}
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
# The original ELF and PS-EXE remain byte/flag unchanged.
& $Objcopy `
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
--set-section-flags ".data=alloc,load,data,code,contents" `
$path_inject_elf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $path_inject_elf"
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
}
else {
Write-Host "[build] DWARF-injected ELF: $path_inject_elf"
}
}
# inject-dwarf
function build-hello_psyqo { function build-hello_psyqo {
$includes += @() $includes += @()
@@ -350,10 +401,10 @@ function build-graphis_hello {
} }
# build-graphis_hello # build-graphis_hello
function build-gte_hello { function build-hello_gte {
$includes += @() $includes += @()
$path_module = join-path $path_code 'gte_hello' $path_module = join-path $path_code 'hello_gte'
$path_duffle = join-path $path_code 'duffle' $path_duffle = join-path $path_code 'duffle'
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h' $path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
$path_build_gen = join-path $path_build 'gen' $path_build_gen = join-path $path_build 'gen'
@@ -402,64 +453,110 @@ function build-gte_hello {
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start). # Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf) ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
$dwarfLineBin = join-path $path_build_gen 'hello_gte.dwarf_line.bin' inject-dwarf $elf $path_build_gen
$dwarfArangesBin = join-path $path_build_gen 'hello_gte.dwarf_aranges.bin'
$dwarfRnglistsBin = join-path $path_build_gen 'hello_gte.dwarf_rnglists.bin'
$injectElf = join-path $path_build 'hello_gte.dwarf-injected.elf'
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
{
Write-Host "[build] DWARF-injecting $elf -> $injectElf"
Copy-Item -LiteralPath $elf -Destination $injectElf -Force
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
$f_args = @(
"--update-section=.debug_line=$dwarfLineBin",
"--update-section=.debug_aranges=$dwarfArangesBin",
"--update-section=.debug_rnglists=$dwarfRnglistsBin"
)
& $Objcopy @f_args $injectElf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] objcopy F' splice failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
return;
}
$dwarfInfoBin = join-path $path_build_gen 'hello_gte.dwarf_info.bin'
$dwarfAbbrevBin = join-path $path_build_gen 'hello_gte.dwarf_abbrev.bin'
$dwarfStrBin = join-path $path_build_gen 'hello_gte.dwarf_str.bin'
$dwarfLocBin = join-path $path_build_gen 'hello_gte.dwarf_loc.bin'
$dwarfLoclistsBin = join-path $path_build_gen 'hello_gte.dwarf_loclists.bin'
$g_args = @(
"--update-section=.debug_info=$dwarfInfoBin",
"--update-section=.debug_abbrev=$dwarfAbbrevBin",
"--update-section=.debug_str=$dwarfStrBin",
"--add-section=.debug_loc=$dwarfLocBin",
"--add-section=.debug_loclists=$dwarfLoclistsBin"
)
& $Objcopy @g_args $injectElf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] objcopy G' splice failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
return;
}
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
# The original ELF and PS-EXE remain byte/flag unchanged.
& $Objcopy `
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
--set-section-flags ".data=alloc,load,data,code,contents" `
$injectElf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
}
else {
Write-Host "[build] DWARF-injected ELF: $injectElf"
}
}
} }
build-gte_hello # build-hello_gte
function build-hello_joypad {
$includes += @()
$path_module = join-path $path_code 'hello_joypad'
$path_duffle = join-path $path_code 'duffle'
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
$path_build_gen = join-path $path_build 'gen'
$src_c = join-path $path_module 'hello_joypad.c'
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
$assemble_args = @()
$assemble_args += $f_debug
$assemble_args += $f_optimize_none
$assemble_args += ($f_include + $path_code)
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
$module_asm_crt = join-path $path_build 'crt0.o'
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
$module_c = join-path $path_build 'hello_joypad_c.o'
$compile_args = @()
$compile_args += $f_debug
$compile_args += $f_optimize_none
# $compile_args += $f_optimize_intrinsics
# $compile_args += $f_optimize_size
# $compile_args += $f_optimize_debug
$compile_args += ($f_include + $path_code)
compile-unit $src_c $module_c $includes $compile_args
$elf = join-path $path_build 'hello_joypad.elf'
$exe = join-path $path_build 'hello_joypad.ps-exe'
$link_args = @()
$link_args += $f_debug
# $link_args += $f_optimize_size
$link_modules = @(
$module_asm_crt,
$module_c
)
link-modules $link_modules $elf $link_args
make-binary $elf $exe
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
inject-dwarf $elf $path_build_gen
}
# build-hello_joypad
function build-hello_camera {
$includes += @()
$path_module = join-path $path_code 'hello_camera'
$path_duffle = join-path $path_code 'duffle'
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
$path_build_gen = join-path $path_build 'gen'
$src_c = join-path $path_module 'hello_camera.c'
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
$assemble_args = @()
$assemble_args += $f_debug
$assemble_args += $f_optimize_none
$assemble_args += ($f_include + $path_code)
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
$module_asm_crt = join-path $path_build 'crt0.o'
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
$module_c = join-path $path_build 'hello_camera_c.o'
$compile_args = @()
$compile_args += $f_debug
$compile_args += $f_optimize_none
# $compile_args += $f_optimize_intrinsics
# $compile_args += $f_optimize_size
# $compile_args += $f_optimize_debug
$compile_args += ($f_include + $path_code)
compile-unit $src_c $module_c $includes $compile_args
$elf = join-path $path_build 'hello_camera.elf'
$exe = join-path $path_build 'hello_camera.ps-exe'
$link_args = @()
$link_args += $f_debug
# $link_args += $f_optimize_size
$link_modules = @(
$module_asm_crt,
$module_c
)
link-modules $link_modules $elf $link_args
make-binary $elf $exe
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
inject-dwarf $elf $path_build_gen
}
build-hello_camera
# NO idea if this works yet... # NO idea if this works yet...
function Send-ToEmulator { param( [string]$exePath ) function Send-ToEmulator { param( [string]$exePath )
+331 -259
View File
File diff suppressed because it is too large Load Diff
+2 -3
View File
@@ -47,15 +47,14 @@ local function find_repo_root()
return root return root
end end
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and --- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
--- `package.cpath` (for `lpeg.dll`).
--- ---
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods. --- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs). --- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, --- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state). --- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
function M.setup() function M.setup()
local repo_root = find_repo_root() local repo_root = find_repo_root()
if not repo_root then if not repo_root then
-- Unreachable in practice: find_repo_root() derives the repo root from this script's -- Unreachable in practice: find_repo_root() derives the repo root from this script's
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms). -- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
+418
View File
@@ -0,0 +1,418 @@
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
-- both parsed ELF32 headers from wire bytes.
--
-- This module contains the format constants and the byte-level walker.
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
--
-- **Adapter contract (explicit pass style):**
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
-- so callers wrap once in a 1-line adapter that strips `self`.
-- The parsers here operate on the unwrapped form.
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
-- read_u8(adapter, off) -> integer | nil
-- read_u16(adapter, off) -> integer | nil
-- read_u32(adapter, off) -> integer | nil
-- size(adapter) -> integer
--
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
--
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
local M = {}
-- ════════════════════════════════════════════════════════════════════════════
-- Little-endian readers (bit-weighted accumulator, math.floor only)
-- ════════════════════════════════════════════════════════════════════════════
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
---
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
--- byte 0 contributes its value directly;
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
---
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
---
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u32(adapter, off)
return adapter.read_u8_at(off)
+ adapter.read_u8_at(off + 0x01) * 0x00000100
+ adapter.read_u8_at(off + 0x02) * 0x00010000
+ adapter.read_u8_at(off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u16(adapter, off)
return adapter.read_u8_at(off)
+ adapter.read_u8_at(off + 0x01) * 0x00000100
end
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u8(adapter, off)
return adapter.read_u8_at(off)
end
--- Total adapter byte length.
--- @param adapter table
--- @return integer
function M.size(adapter)
return adapter.read_size()
end
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
--- both layers now use the same byte-level helpers under the hood.
function M.read_u32_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off)
+ buf:byte(byte_off + 0x01) * 0x00000100
+ buf:byte(byte_off + 0x02) * 0x00010000
+ buf:byte(byte_off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- @param buf string
--- @param off integer -- zero-based wire offset
--- @return integer
function M.read_u16_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
end
-- ════════════════════════════════════════════════════════════════════════════
-- Format constants
-- ════════════════════════════════════════════════════════════════════════════
-- ELF format constants (System V ABI gABI v1.2).
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
-- ---------------------------------------------------------------------------
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
-- ---------------------------------------------------------------------------
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
M.ELF32_HEADER = {
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
magic = "\127ELF",
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
header_bytes = 0x34, -- ELF32 header is 52 bytes total
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
}
-- ---------------------------------------------------------------------------
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
-- ---------------------------------------------------------------------------
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
-- zero-based, field offsets relative to the start of the entry.
M.ELF32_SECTION = {
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
}
-- ---------------------------------------------------------------------------
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
-- ---------------------------------------------------------------------------
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
-- zero-based, field offsets relative to the start of the entry.
M.ELF32_SYM = {
st_name = 0x00, -- 4-byte LE; offset into the linked string table
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
st_size = 0x08, -- 4-byte LE; symbol size in bytes
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
}
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
M.dw_dwarf32_terminator = 0xFFFFFFFF
-- ════════════════════════════════════════════════════════════════════════════
-- Adapter validation
-- ════════════════════════════════════════════════════════════════════════════
--- Validate that `adapter` exposes the byte-read surface.
--- Returns true on success, false + a stable error code on failure.
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
--- @param adapter any
--- @return boolean, string|nil
function M.validate_adapter(adapter)
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
return true, nil
end
-- ════════════════════════════════════════════════════════════════════════════
-- String-table reader
-- ════════════════════════════════════════════════════════════════════════════
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
--- @param strtab string
--- @param off integer
--- @return string|nil
function M.get_str(strtab, off)
if off < 0 or off >= #strtab then return nil end
local end_pos = strtab:find("\0", off + 1, true)
if not end_pos then return nil end
return strtab:sub(off + 1, end_pos - 1)
end
-- ════════════════════════════════════════════════════════════════════════════
-- Header / section / symbol walkers
-- ════════════════════════════════════════════════════════════════════════════
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
--- Returns a table on success:
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
--- On failure returns nil + a stable error code:
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
--- @param adapter table
--- @return table|nil, string|nil
function M.parse_elf32_headers(adapter)
local ok, err = M.validate_adapter(adapter)
if not ok then return nil, err end
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
-- The byte readers take the adapter explicitly.
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
local b1 = M.read_u8(adapter, 0)
local b2 = M.read_u8(adapter, 1)
local b3 = M.read_u8(adapter, 2)
local b4 = M.read_u8(adapter, 3)
if not (b1 and b2 and b3 and b4)
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
return nil, "bad_magic"
end
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset)
if class ~= M.ELFCLASS32 then
return nil, "unsupported_elf_class"
end
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset)
if data ~= M.ELFDATA2LSB then
return nil, "unsupported_elf_data"
end
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset)
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset)
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset)
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset)
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset)
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
return nil, "truncated_header"
end
return {
e_entry = e_entry,
e_shoff = e_shoff,
e_shentsize = e_shentsize,
e_shnum = e_shnum,
e_shstrndx = e_shstrndx,
error = nil,
}
end
--- Read one section-header entry from `adapter` at `sh_off`.
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
--- @param adapter table
--- @param sh_off integer
--- @return table|nil, string|nil -- entry, error
local function read_section_entry(adapter, sh_off)
local entry = {
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
name = "",
}
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
and entry.sh_offset and entry.sh_size and entry.sh_link) then
return nil, "truncated_section_headers"
end
return entry, nil
end
--- Walk every section header in `hdr` and return a 1-based array of entries
--- (the section at logical index 0 is at array position 1, etc.).
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
--- @param adapter table
--- @param hdr table -- the table returned by parse_elf32_headers
--- @return table|nil, string|nil
function M.walk_sections(adapter, hdr)
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
local file_size = M.size(adapter)
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
return nil, "truncated_section_headers"
end
-- Read every section header first; we need .shstrtab to resolve names.
local sections = {}
for i = 0, hdr.e_shnum - 1 do
local sh_off = hdr.e_shoff + i * hdr.e_shentsize
local entry, err = read_section_entry(adapter, sh_off)
if not entry then return nil, err end
sections[i + 1] = entry
end
if hdr.e_shstrndx >= hdr.e_shnum then
return nil, "missing_shstrtab"
end
local shstrtab = sections[hdr.e_shstrndx + 1]
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
return nil, "missing_shstrtab"
end
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
return nil, "truncated_section_headers"
end
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab)
if not shstrtab_bytes then return nil, "truncated_section_headers" end
for _, s in ipairs(sections) do
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
end
return sections, nil
end
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
--- @param adapter table
--- @param section table -- one entry from walk_sections
--- @return string|nil
function M.read_section_bytes(adapter, section)
local size = section.sh_size
if size == 0 then return "" end
local out = {}
for i = 0, size - 1 do
local b = M.read_u8(adapter, section.sh_offset + i)
if b == nil then return nil end
out[#out + 1] = string.char(b)
end
return table.concat(out)
end
--- Convenience: walk sections, then look up the named section, then read its bytes.
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
--- @param adapter table
--- @param sections table -- 1-based array from walk_sections
--- @param name string
--- @return string|nil, string|nil
function M.read_named_section(adapter, sections, name)
if not sections then return nil, "missing_section" end
for _, s in ipairs(sections) do
if s.name == name then
local bytes = M.read_section_bytes(adapter, s)
if not bytes then return nil, "truncated_section_data" end
return bytes, nil
end
end
return nil, "missing_section"
end
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
--- @param adapter table
--- @param sections table
--- @return table|nil, string|nil
function M.collect_symbols(adapter, sections)
if not sections then return nil, "missing_sections" end
local symbols = {}
local file_size = M.size(adapter)
for _, s in ipairs(sections) do
if s.sh_type == M.SHT_SYMTAB then
local strtab = sections[s.sh_link + 1]
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
return nil, "missing_symtab_strtab"
end
if strtab.sh_offset + strtab.sh_size > file_size then
return nil, "truncated_section_headers"
end
local strtab_bytes = M.read_section_bytes(adapter, strtab)
if not strtab_bytes then return nil, "truncated_section_headers" end
if s.sh_offset + s.sh_size > file_size then
return nil, "truncated_section_headers"
end
local symtab_bytes = M.read_section_bytes(adapter, s)
if not symtab_bytes then return nil, "truncated_section_headers" end
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes
for j = 0, n - 1 do
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name)
if st_name then
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value)
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size)
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info)
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
-- the metaprogram reads too. Inline the read to keep the
-- adapter as the only I/O surface.
local b1 = M.read_u8(adapter, e + 14)
local b2 = M.read_u8(adapter, e + 15)
if not (b1 and b2) then
return nil, "truncated_section_headers"
end
local st_shndx = b1 + b2 * 0x100
local name = M.get_str(strtab_bytes, st_name) or ""
if name ~= "" then
symbols[name] = {
value = st_value,
size = st_size,
info = st_info,
shndx = st_shndx,
}
end
end
end
end
end
return symbols, nil
end
return M
+440 -182
View File
@@ -1,13 +1,8 @@
--- elf_dwarf.lua — ELF32 + DWARF + atoms source-map utilities. --- elf_dwarf.lua — ELF32 + DWARF + atoms source-map utilities.
--- All ELF32 + DWARF-specific code lives here.
---
--- **What this module contains:** --- **What this module contains:**
--- - **Format-constant tables** (the byte-offset / opcode / size encyclopedias for ELF32, DWARF4 aranges, DWARF5 rnglists, DWARF line-program, MIPS). --- - **Format-constant tables** (the byte-offset / opcode / size encyclopedias for ELF32, DWARF4 aranges, DWARF5 rnglists, DWARF line-program, MIPS).
--- Every constant carries a spec:` comment naming the spec section that defines it. --- Every constant carries a spec:` comment naming the spec section that defines it.
--- - **I/O helpers**: little-endian byte read/write, ELF32 section walker, nm symbol reader, source-map parser, native directory glob. --- - **I/O helpers**: little-endian byte read/write, ELF32 section walker, nm symbol reader, source-map parser, native directory glob.
---
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Native dependencies -- Native dependencies
@@ -16,6 +11,11 @@
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`). -- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
local lfs = require("lfs") local lfs = require("lfs")
-- scripts/elf32.lua contains format-constant tables + the byte-level walker.
-- The this file re-exports `read_u32_le` / `read_u16_le` (and the DWARF32 terminator).
-- TODO(Ed): Remove re-export.
local E = require("elf32")
local M = {} local M = {}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -60,19 +60,19 @@ M.DW_AT = {
} }
M.DW_FORM = { M.DW_FORM = {
addr = 0x01, addr = 0x01,
data1 = 0x0B, data1 = 0x0B,
data2 = 0x05, data2 = 0x05,
data4 = 0x06, data4 = 0x06,
string = 0x08, string = 0x08,
strp = 0x0E, strp = 0x0E,
exprloc = 0x18, exprloc = 0x18,
ref4 = 0x13, ref4 = 0x13,
udata = 0x0F, udata = 0x0F,
ref_sig8 = 0x20, ref_sig8 = 0x20,
implicit_const = 0x21, implicit_const = 0x21,
flag_present = 0x19, flag_present = 0x19,
sec_offset = 0x17, sec_offset = 0x17,
} }
M.DW_ATE = { M.DW_ATE = {
@@ -104,34 +104,16 @@ M.MIPS_BYTES_PER_WORD = 0x04
-- ---------------------------------------------------------------------------- -- ----------------------------------------------------------------------------
-- ELF32 (System V ABI gABI v1.2) -- ELF32 (System V ABI gABI v1.2)
-- ---------------------------------------------------------------------------- -- ----------------------------------------------------------------------------
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, --- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
--- and section-relative values are zero-based wire offsets. Only Lua string APIs receive --- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
--- a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`). --- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
--- ---
--- ELF/DWARF field offsets are expressed in hex so they map directly to the --- The ELF32 header / section / sym layout tables are within scripts/elf32.lua.
--- zero-based byte positions in the binary file. --- The metaprogram re-exports the DWARF32 initial-length terminator.
--- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table" M.dw_dwarf32_terminator = E.dw_dwarf32_terminator
M.ELF32 = { -- TODO(Ed): Remove re-export.
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00
magic = "\127ELF",
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
class_elf32 = 1,
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
endian_little = 1,
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
}
-- ---------------------------------------------------------------------------- -- ----------------------------------------------------------------------------
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table) -- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
@@ -192,8 +174,7 @@ M.DWARF_LINE_OPS = {
DW_LNE_end_sequence = 1, -- spec: §6.2.5.3 DW_LNE_end_sequence = 1, -- spec: §6.2.5.3
DW_LNE_set_address = 2, -- spec: §6.2.5.3 DW_LNE_set_address = 2, -- spec: §6.2.5.3
-- Standard opcode header (§6.2.5.1) -- Standard opcode header (§6.2.5.1)
-- opcode_base + line_range are 1-byte header fields; hex so they map -- opcode_base + line_range are 1-byte header fields; hex so they map directly to the line-program header byte sequence.
-- directly to their position in the line-program header byte sequence.
-- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics. -- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics.
opcode_base = 0x0D, opcode_base = 0x0D,
line_base = -5, line_base = -5,
@@ -204,6 +185,44 @@ M.DWARF_LINE_OPS = {
set_address_payload_size = 0x05, -- size = sub_opcode(1) + addr(4) set_address_payload_size = 0x05, -- size = sub_opcode(1) + addr(4)
} }
-- ----------------------------------------------------------------------------
-- DWARF5 .debug_line (per DWARF5 spec §6.2.4 — Line Number Program Header)
-- ----------------------------------------------------------------------------
-- All offsets are zero-based wire offsets from the start of the unit body
-- (i.e. AFTER unit_length has been read and unit_length bytes skipped past unit_length's 4 bytes).
--
-- The DWARF3/4 line-program format differs:
-- - It omits `address_size` (DWARF3 §6.2.4) + `segment_selector_size` (DWARF5 §6.2.4).
-- - It uses null-terminated string lists for `include_directories` + `file_names`
-- (vs. DWARF5's format_count + fields-list shape).
-- These are documented inline at each parse site in read_line_unit_file_table below.
--- spec: DWARF5 spec §6.2.4 (Line Number Program Header — version >= 5)
M.DWARF5_DEBUG_LINE = {
-- Header fields (zero-based, AFTER unit_length has been read).
version_offset_post_il = 0x00, -- 2-byte LE; expected = 5
addr_size_offset = 0x02, -- 1 byte; expected = 4
seg_size_offset = 0x03, -- 1 byte; expected = 0
header_length_offset = 0x04, -- 4-byte LE; length of program-header content that follows
program_header_start = 0x08, -- first byte of program-header content (after the 8 fixed bytes)
-- Per-form byte widths (used when reading directory / file-name entries).
form_addr_bytes = 0x04, -- DW_FORM_addr (32-bit) | DW_FORM_data4
form_strp_bytes = 0x04, -- DW_FORM_line_strp / DW_FORM_strp / DW_FORM_strp_sup
form_data16_bytes = 0x10, -- DW_FORM_data16 (MD5)
-- DWARF5 form codes (subset used in line-program directory + file tables).
form_line_strp = 0x1A, -- DWARF5 §7.5.6 — DW_FORM_line_strp (4-byte offset into .debug_line_str)
form_string = 0x08, -- DWARF4-compatible fallback (inline null-terminated; not in .debug_line_str)
form_udata = 0x0F, -- DW_FORM_udata (ULEB)
form_data16 = 0x18, -- DW_FORM_data16 (16-byte MD5; gcc emits this for split debug info)
-- DWARF5 content-tag codes (DW_LNCT_* from §6.2.4.1 + §6.2.4.2).
lnct_path = 0x01,
lnct_directory_index = 0x02,
lnct_md5 = 0x05, -- gcc with MD5 in file name table (rare)
}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- I/O helpers: little-endian byte read/write -- I/O helpers: little-endian byte read/write
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -211,41 +230,35 @@ M.DWARF_LINE_OPS = {
--- Read a 4-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`. --- Read a 4-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- Equivalent to `string.unpack("<I4", buf, off + 1)` but avoids the table-return shape + works under LuaJIT 2.1 --- Equivalent to `string.unpack("<I4", buf, off + 1)` but avoids the table-return shape + works under LuaJIT 2.1
--- (which has partial `string.unpack` coverage). --- (which has partial `string.unpack` coverage).
---
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary. --- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
--- ---
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit: --- Thin forwarder: the canonical implementation lives in scripts/elf32.lua.
--- byte 0 contributes its value directly; byte 1 is shifted left by 8 --- The "second caller lifts" pattern keeps the metaprogram side fluent
--- (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000). --- (`M.read_u32_le(buf, off)`) while the body is deduped.
--- @param buf string --- @param buf string
--- @param off integer -- zero-based wire offset --- @param off integer -- zero-based wire offset
--- @return integer --- @return integer
function M.read_u32_le(buf, off) function M.read_u32_le(buf, off)
local byte_off = off + 1 return E.read_u32_le(buf, off)
return buf:byte(byte_off)
+ buf:byte(byte_off + 0x01) * 0x00000100
+ buf:byte(byte_off + 0x02) * 0x00010000
+ buf:byte(byte_off + 0x03) * 0x01000000
end end
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`. --- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.) --- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
--- Thin forwarder — see `M.read_u32_le` for the rationale.
--- @param buf string --- @param buf string
--- @param off integer -- zero-based wire offset --- @param off integer -- zero-based wire offset
--- @return integer --- @return integer
function M.read_u16_le(buf, off) function M.read_u16_le(buf, off)
local byte_off = off + 1 return E.read_u16_le(buf, off)
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
end end
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser. -- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
-- Offsets are 0-based; returns (value, next_pos). -- Offsets are 0-based; returns (value, next_pos).
-- Promoted from `local function` to M.* exports so passes/dwarf_injection.lua -- Promoted from `local function` to M.* exports so passes/dwarf_injection.lua can import them as file-scope locals per the 2nd-caller lift precedent
-- can import them as file-scope locals per the 2nd-caller lift precedent
-- (the uleb128 + sleb128 encoders were promoted the same way). -- (the uleb128 + sleb128 encoders were promoted the same way).
function M.read_uleb128_at(buf, pos) function M.read_uleb128_at(buf, pos)
local value, shift = 0, 0 local value, shift = 0, 0
local len = #buf local len = #buf
while pos < len do while pos < len do
local b = buf:byte(pos + 1) local b = buf:byte(pos + 1)
value = value + (b % 0x80) * (2 ^ shift) value = value + (b % 0x80) * (2 ^ shift)
@@ -389,13 +402,13 @@ local function read_form_value(buf, str_buf, pos, form)
-- The constant is declared in the abbrev; no value bytes in the DIE. -- The constant is declared in the abbrev; no value bytes in the DIE.
return nil, pos return nil, pos
elseif form == M.DW_FORM.ref_sig8 then elseif form == M.DW_FORM.ref_sig8 then
-- DW_FORM_ref_sig8 (DWARF5 §7.4.2): an 8-byte value identifying a type -- DW_FORM_ref_sig8 (DWARF5 §7.4.2): An 8-byte value identifying a type by signature.
-- by signature. The low 4 bytes (LE) are the type signature (content hash); -- The low 4 bytes (LE) are the type signature (content hash);
-- the high 4 bytes (LE) are a CU-relative offset into the matching type unit. -- The high 4 bytes (LE) are a CU-relative offset into the matching type unit.
-- Consumers use the low 4 to look up the type unit (see M.find_type_unit_by_signature) -- Consumers use the low 4 to look up the type unit (see M.find_type_unit_by_signature)
-- then the high 4 to resolve the specific type within it. -- then the high 4 to resolve the specific type within it.
-- Return the low 4 as the primary value to preserve the (value, next_pos) shape; -- Return the low 4 as the primary value to preserve the (value, next_pos) shape;
-- the high 4 is exposed via M.read_ref_sig8 (which returns both halves). -- the high 4 is exposed via M.read_ref_sig8 (which returns both halves).
local _, _, next_pos = M.read_ref_sig8(buf, pos) local _, _, next_pos = M.read_ref_sig8(buf, pos)
return M.read_u32_le(buf, pos), next_pos return M.read_u32_le(buf, pos), next_pos
else else
@@ -417,26 +430,26 @@ function M.read_ref_sig8(buf, pos)
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8 return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
end end
-- DWARF5 §7.5.6 (Type Entries). --- DWARF5 §7.5.6 (Type Entries).
-- Walk all units in `info` and return the 0-based offset of the first unit --- Walk all units in `info` and return the 0-based offset of the first unit whose `DW_AT_type_signature`
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`. --- (8-byte value at the end of the unit header) equals `target_sig`.
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract; --- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists. --- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
-- ---
-- Unit header layout (from pos 0): --- Unit header layout (from pos 0):
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4) --- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4) --- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset). --- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
-- @param info string -- the .debug_info section bytes --- @param info string -- the .debug_info section bytes
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature --- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature --- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit --- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi) function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
local pos = 0 local pos = 0
local section_len = #info local section_len = #info
while pos + 4 < section_len do while pos + 4 < section_len do
local unit_length = M.read_u32_le(info, pos) local unit_length = M.read_u32_le(info, pos)
if unit_length == 0xFFFFFFFF then if unit_length == 0xFFFFFFFF then
return nil, nil -- DWARF64 not supported return nil, nil -- DWARF64 not supported
end end
-- unit_length is the body size, NOT including the 4-byte unit_length field itself. -- unit_length is the body size, NOT including the 4-byte unit_length field itself.
@@ -465,9 +478,9 @@ function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
-- byte 8-15: type_signature (8) -- byte 8-15: type_signature (8)
-- byte 16-19: type_offset (4) -- byte 16-19: type_offset (4)
local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed
if unit_type == 0x02 then -- DW_UT_type if unit_type == 0x02 then -- DW_UT_type
local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed
if sig_lo == target_sig_lo and sig_hi == target_sig_hi then if sig_lo == target_sig_lo and sig_hi == target_sig_hi then
local type_offset = M.read_u32_le(info, body_start + 16) -- 0-based +16 = type_offset in 1-indexed local type_offset = M.read_u32_le(info, body_start + 16) -- 0-based +16 = type_offset in 1-indexed
return pos, type_offset return pos, type_offset
end end
@@ -514,7 +527,7 @@ end
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first). --- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
--- For frequent callers, pass the union of all needed sections in one call. --- For frequent callers, pass the union of all needed sections in one call.
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker. -- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
--- @param elf_path Path --- @param elf_path Path
--- @param section_names string[] -- list of section names to read --- @param section_names string[] -- list of section names to read
--- @return table<string, string> --- @return table<string, string>
function M.read_elf_sections(elf_path, section_names) function M.read_elf_sections(elf_path, section_names)
@@ -533,75 +546,64 @@ function M.read_elf_sections(elf_path, section_names)
return result return result
end end
local f = io.open(elf_path, "rb") local f = io.open(elf_path, "rb")
if not f then if not f then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] io.open failed: %s\n", elf_path)) io.stderr:write(string.format("[elf_dwarf.read_elf_sections] io.open failed: %s\n", elf_path))
return result return result
end end
-- Read the ELF32 header. local file_size
local header = f:read(M.ELF32.header_bytes) do
if not header or #header < M.ELF32.header_bytes then f:seek("end", 0)
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n") file_size = f:seek("cur", 0)
end
local adapter = {
read_u8_at = function(offset)
f:seek("set", offset)
local b = f:read(1)
if not b then return nil end
return b:byte()
end,
read_u16_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
if not b1 or not b2 then return nil end
return b1:byte() + b2:byte() * 0x100
end,
read_u32_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
local b3 = f:read(1)
local b4 = f:read(1)
if not b1 or not b2 or not b3 or not b4 then return nil end
return b1:byte() + b2:byte() * 0x100
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
end,
read_size = function() return file_size end,
}
-- Delegate the header parse + section walk to E.*.
local hdr, hdr_err = E.parse_elf32_headers(adapter)
if not hdr then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] header parse failed: %s\n", tostring(hdr_err)))
f:close() f:close()
return result return result
end end
-- Sanity-check magic + class + endianness. local sections, walk_err = E.walk_sections(adapter, hdr)
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then if not sections then
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n") io.stderr:write(string.format("[elf_dwarf.read_elf_sections] section walk failed: %s\n", tostring(walk_err)))
f:close()
return result
end
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
f:close()
return result
end
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
f:close() f:close()
return result return result
end end
-- Parse section-header table location + dimensions from the header. -- Resolve the requested sections.
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset) for _, s in ipairs(sections) do
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset) if wanted[s.name] then
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset) local bytes = E.read_section_bytes(adapter, s)
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset) if bytes then result[s.name] = bytes end
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
local strtab_hdr = f:read(e_shentsize)
if not strtab_hdr or #strtab_hdr < e_shentsize then
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
f:close()
return result
end
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
f:seek("set", strtab_offset)
local strtab = f:read(strtab_size) or ""
-- Walk all section headers; collect (offset, size) for the wanted names.
local function read_section_bytes(sh_offset, sh_size)
f:seek("set", sh_offset)
return f:read(sh_size) or ""
end
for sh_idx = 0, e_shnum - 1 do
f:seek("set", e_shoff + sh_idx * e_shentsize)
local sh = f:read(e_shentsize)
if not sh or #sh < e_shentsize then break end
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
-- Extract the name (null-terminated C string in strtab).
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
local name = strtab:sub(sh_name + 1, name_end - 1)
if wanted[name] then
result[name] = read_section_bytes(sh_offset, sh_size)
end end
end end
@@ -618,49 +620,87 @@ end
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded. --- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix). --- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
--- - `st_size > 0` filter excludes undefined/imported symbols. --- - `st_size > 0` filter excludes undefined/imported symbols.
---
--- @param elf_path Path --- @param elf_path Path
--- @return table<string, {integer, integer}> --- @return table<string, {integer, integer}>
function M.read_nm(elf_path) function M.read_nm(elf_path)
local addrs = {} local addrs = {}
-- Read .symtab + .strtab via the existing ELF walker (no subprocess). -- Existence check first; an empty or missing ELF returns an empty map.
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"}) if lfs.attributes(elf_path, "mode") ~= "file" then
local symtab = sections[".symtab"]
local strtab = sections[".strtab"]
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
-- No symbol table (e.g. stripped ELF). Return empty.
return addrs return addrs
end end
-- Iterate the 16-byte ELF32 symtab entries. local f = io.open(elf_path, "rb")
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14. if not f then
local SYM_ENTRY_BYTES = 0x10 return addrs
local SYM_ST_NAME = 0x00 end
local SYM_ST_VALUE = 0x04
local SYM_ST_SIZE = 0x08 -- Build the file adapter for E.*.
local SYM_ST_INFO = 0x0C local file_size
local n_syms = #symtab / SYM_ENTRY_BYTES do
for i = 0, n_syms - 1 do f:seek("end", 0)
local entry_off = i * SYM_ENTRY_BYTES file_size = f:seek("cur", 0)
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1) end
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2). local adapter = {
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat read_u8_at = function(offset)
-- (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions). f:seek("set", offset)
local binding = math.floor(st_info / 16) local b = f:read(1)
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL if not b then return nil end
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE) return b:byte()
if st_size > 0 then end,
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME) read_u16_at = function(offset)
-- Extract the name from .strtab (null-terminated C string). f:seek("set", offset)
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1) local b1 = f:read(1)
local name = strtab:sub(st_name_off + 1, name_end - 1) local b2 = f:read(1)
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix). if not b1 or not b2 then return nil end
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref. return b1:byte() + b2:byte() * 0x100
if name and #name > 0 then end,
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE) read_u32_at = function(offset)
addrs[name] = { st_value, st_size } f:seek("set", offset)
end local b1 = f:read(1)
end local b2 = f:read(1)
local b3 = f:read(1)
local b4 = f:read(1)
if not b1 or not b2 or not b3 or not b4 then return nil end
return b1:byte() + b2:byte() * 0x100
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
end,
read_size = function() return file_size end,
}
-- Delegate the header + section walk to E.*.
local hdr, hdr_err = E.parse_elf32_headers(adapter)
if not hdr then
io.stderr:write(string.format("[elf_dwarf.read_nm] header parse failed: %s\n", tostring(hdr_err)))
f:close()
return addrs
end
local sections, walk_err = E.walk_sections(adapter, hdr)
if not sections then
io.stderr:write(string.format("[elf_dwarf.read_nm] section walk failed: %s\n", tostring(walk_err)))
f:close()
return addrs
end
-- E.collect_symbols returns every defined symbol (no binding filter).
-- The metaprogram then applies its STB_LOCAL / STB_GLOBAL + size>0 filter, matching `nm`'s default (external symbols only).
local symbols, sym_err = E.collect_symbols(adapter, sections)
if not symbols then
io.stderr:write(string.format("[elf_dwarf.read_nm] symbol collection failed: %s\n", tostring(sym_err)))
f:close()
return addrs
end
f:close()
for name, entry in pairs(symbols) do
-- High nibble of st_info = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
-- math.floor(/16) is portable across LuaJIT 2.0/2.1 and plain Lua 5.x.
local binding = math.floor(entry.info / 16)
if (binding == 0 or binding == 1) and entry.size > 0 then
addrs[name] = { entry.value, entry.size }
end end
end end
@@ -741,8 +781,8 @@ function M.sleb128(n)
local b = n % (LEB_DATA_MASK + 1) -- extract low 7 bits local b = n % (LEB_DATA_MASK + 1) -- extract low 7 bits
n = (n - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7 n = (n - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
-- Termination: remaining value bits fit in the sign bit of the last byte. -- Termination: remaining value bits fit in the sign bit of the last byte.
if n == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator if n == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
if n == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator if n == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
if more then b = b + LEB_CONT_BIT end if more then b = b + LEB_CONT_BIT end
bytes[#bytes + 1] = string.char(b) bytes[#bytes + 1] = string.char(b)
end end
@@ -773,20 +813,238 @@ end
--- @param n integer -- any integer (negative allowed) --- @param n integer -- any integer (negative allowed)
--- @return integer --- @return integer
function M.sleb128_size(n) function M.sleb128_size(n)
local more = true local more = true
local bytes = 0 local bytes = 0
local v = n local v = n
while more do while more do
local b = v % (LEB_DATA_MASK + 1) -- extract low 7 bits local b = v % (LEB_DATA_MASK + 1) -- extract low 7 bits
v = (v - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7 v = (v - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
if v == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator if v == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
if v == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator if v == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
if more then b = b + LEB_CONT_BIT end if more then b = b + LEB_CONT_BIT end
bytes = bytes + 1 bytes = bytes + 1
end end
return bytes return bytes
end end
-- ════════════════════════════════════════════════════════════════════════════
-- DWARF5 line-program file-table reader
-- ════════════════════════════════════════════
--- Read every line-program unit in `.debug_line` and produce one entry per file across all units.
--- Returns three parallel maps keyed by 1-based file index.
---
--- Wire format notes:
--- * The `.debug_line` section may contain MULTIPLE line-program units
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom `inv.call_file`
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` emits 2 forms: path + dir_index).
--- The helper supports:
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
--- - DW_FORM_udata (ULEB128)
--- - DW_FORM_data16 (16-byte MD5; ignored — skip the form's bytes)
--- * Symlink-canonicalisation: each path's `paths[i]` is stored verbatim from the wire
--- (mixed `/` and `\` accepted; the basename is taken via the last path separator). Caller normalises as needed.
---
--- Behavior on failure: writes to stderr and returns nil.
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
--- downstream `resolve_provenance_file_index(path)` consumers consult the map directly.
---
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
--- @return table|nil, table|nil, table|nil
--- basename_to_index: { [basename] = 1-based-per-unit-file-index, ... }
--- basenames: { [1-based-per-unit-file-index] = basename, ... }
--- paths: { [1-based-per-unit-file-index] = full path (mixed slashes), ... }
function M.read_line_unit_file_table(elf_path)
local sections = M.read_elf_sections(elf_path, { ".debug_line", ".debug_line_str" })
local line = sections[".debug_line"]
local lstr = sections[".debug_line_str"] or ""
if not line or line == "" then
io.stderr:write("[elf_dwarf.read_line_unit_file_table] no .debug_line section in: " .. tostring(elf_path) .. "\n")
return nil
end
local basenames = {}
local basename_to_index = {}
local paths = {}
--- Read one form-code's bytes from `buf` at position `p` according to `form`.
--- Returns (value, after) where `value` is:
--- * the resolved string (DW_FORM_line_strp / DW_FORM_string)
--- * the ULEB128 number (DW_FORM_udata)
--- * nil + skip-bytes (DW_FORM_data16; we don't surface the MD5)
local function read_form(buf, lstr_buf, p, form)
if form == M.DWARF5_DEBUG_LINE.form_line_strp then
local strp = M.read_u32_le(buf, p)
local end_pos = lstr_buf:find("\0", strp + 1, true) or (#lstr_buf + 1)
return lstr_buf:sub(strp + 1, end_pos - 1), p + M.DWARF5_DEBUG_LINE.form_strp_bytes
elseif form == M.DWARF5_DEBUG_LINE.form_string then
local nul = buf:find("\0", p + 1, true) or (#buf + 1)
return buf:sub(p + 1, nul - 1), nul
elseif form == M.DWARF5_DEBUG_LINE.form_udata then
local v, after = M.read_uleb128_at(buf, p)
return v, after
elseif form == M.DWARF5_DEBUG_LINE.form_data16 then
return nil, p + M.DWARF5_DEBUG_LINE.form_data16_bytes
else
-- Unsupported form in a directory/file-table entry: best-effort skip.
-- We do NOT stderr-write because the crt0.s DWARF5 line unit (gcc-as emitted) uses DW_FORM_addr (0x01) for what is effectively a path entry, which is non-standard.
-- The C-unit's DWARF3 paths are read via the parallel DWARF3 path and never see this error.
-- Callers should consult `basename_to_index` for the paths they care about and ignore this unit if it produced none.
return nil, p
end
end
--- Parse one DWARF-version-3-style unit (DWARF3/4 line program; gcc default in the PS1 toolchain still emits DWARF3 for line programs in `-g` mode).
--- Layout: null-terminated directory list, then path(null) + dir_idx(ULEB) + time(ULEB) + size(ULEB) file entries terminated by an empty null.
--- `content_start` = zero-based wire offset of the first byte of program-header content (after version + header_length fields).
--- @return unit_basenames { [idx_in_unit_1_based] = basename }
--- @return unit_paths { [idx_in_unit_1_based] = full path }
local function parse_dwarf3_unit(buf, content_start, body_end)
local up = content_start
-- 5 fixed bytes: min_insn, default_is, line_base (signed), line_range, opcode_base
up = up + 5
local opcode_base = buf:byte(content_start + 5)
up = up + (opcode_base - 1) -- std_opcode_lengths
local dirs = {}
while up < body_end do
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
if nul > body_end then break end
local len = nul - up - 1
if len == 0 then up = nul break end
dirs[#dirs + 1] = buf:sub(up + 1, nul - 1)
up = nul
end
local unit_basenames = {}
local unit_paths = {}
while up < body_end do
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
if nul > body_end or nul == up + 1 then up = nul break end
local path = buf:sub(up + 1, nul - 1)
up = nul
local didx, up_next = M.read_uleb128_at(buf, up); up = up_next
local _time, up_next2 = M.read_uleb128_at(buf, up); up = up_next2
local _size, up_next3 = M.read_uleb128_at(buf, up); up = up_next3
local idx = #unit_basenames + 1
local bs = path:match("[^/\\]+$") or path
unit_paths[idx] = path
unit_basenames[idx] = bs
dirs[1] = dirs[1] or "" -- safety: gcc emits "" sentinel dir at 0
if didx > 0 and dirs[didx] then
unit_paths[idx] = dirs[didx] .. "/" .. path
end
end
return unit_basenames, unit_paths
end
--- Parse one DWARF-version-5-style unit (DWARF5 line program; used by modern gcc with `-gdwarf-5`).
--- `content_start` is the first byte of program-header content (after the 8 fixed bytes version+addr_size+seg_size+header_length).
--- @return same shape as parse_dwarf3_unit
local function parse_dwarf5_unit(buf, lstr_buf, content_start, body_end)
local up = content_start
-- 6 fixed bytes: min_insn, max_ops_per_insn, default_is, line_base, line_range, opcode_base
up = up + 6
local opcode_base = buf:byte(content_start + 6)
up = up + (opcode_base - 1) -- std_opcode_lengths
-- directories
local dir_format_count, after = M.read_uleb128_at(buf, up); up = after
local dir_formats = {}
for i = 1, dir_format_count do
local f, a2 = M.read_uleb128_at(buf, up); up = a2
dir_formats[i] = f
end
local dir_count, a3 = M.read_uleb128_at(buf, up); up = a3
local dirs = {}
for i = 1, dir_count do
local combined = ""
for j = 1, dir_format_count do
local v, a4 = read_form(buf, lstr_buf, up, dir_formats[j])
up = a4
if j == 1 and type(v) == "string" then combined = v end
end
dirs[i] = combined
end
-- file names
local file_format_count, after2 = M.read_uleb128_at(buf, up); up = after2
local file_formats = {}
for i = 1, file_format_count do
local f, a2 = M.read_uleb128_at(buf, up); up = a2
file_formats[i] = f
end
local file_count, a3 = M.read_uleb128_at(buf, up); up = a3
local unit_basenames = {}
local unit_paths = {}
for i = 1, file_count do
local combined = ""
local didx = 0
for j = 1, file_format_count do
local v, a4 = read_form(buf, lstr_buf, up, file_formats[j])
up = a4
if j == 1 and type(v) == "string" then combined = v end
if j == 2 and type(v) == "number" then didx = v end
end
local idx = #unit_basenames + 1
local bs = combined:match("[^/\\]+$") or combined
unit_paths[idx] = combined
unit_basenames[idx] = bs
if didx > 0 and dirs[didx] then
unit_paths[idx] = dirs[didx] .. "/" .. combined
end
end
return unit_basenames, unit_paths
end
--- Walk every line-program unit in the section.
local p = 0
local section_end = #line
while p + 4 <= section_end do
local unit_length = M.read_u32_le(line, p)
if unit_length == 0xFFFFFFFF then
io.stderr:write("[elf_dwarf.read_line_unit_file_table] 64-bit DWARF (initial-length 0xFFFFFFFF); not supported\n")
return nil
end
local body_start = p + 4
local body_end = p + 4 + unit_length
if body_end > section_end then break end
local version = M.read_u16_le(line, body_start)
local unit_basenames, unit_paths
if version >= 5 then
-- DWARF5 header: version(2) + addr_size(1) + seg_size(1) + header_length(4) + content
local header_length_offset = body_start + 6 -- past version(2) + addr_size(1) + seg_size(1) - wait that's wrong; past hdr len is at +6
local content_start = body_start + 8 -- past version(2) + addr_size(1) + seg_size(1) + header_length(4)
unit_basenames, unit_paths = parse_dwarf5_unit(line, lstr, content_start, body_end)
elseif version >= 2 then
-- DWARF2/3/4 header: version(2) + header_length(4) + content
local content_start = body_start + 6 -- past version(2) + header_length(4)
unit_basenames, unit_paths = parse_dwarf3_unit(line, content_start, body_end)
else
io.stderr:write(string.format("[elf_dwarf.read_line_unit_file_table] unsupported DWARF version %d (offset 0x%x)\n", version, p))
p = body_end
goto continue
end
-- Per-unit 1-based file indices are aligned with `inv.call_file` values because the metaprogram emits `DW_LNS_set_file` with the per-unit index.
-- When multiple units are present (crt0.s + C unit), the per-unit index in each unit matches the metaprogram's intent (gcc always sets file in unit-local terms).
-- We therefore store directly without global re-indexing; the caller is responsible for knowing which unit the file-index applies to.
-- For DWARF3 (C unit is the unit that matters for atom line tables), this matches.
-- For DWARF5 (crt0.s + C unit), each carries its own per-unit file-table map;
-- the atom-side DW_LNS_set_file(N) refers to the C unit's indices, NOT crt0.s's.
-- Since the C unit is the one with full include_directories + 12 entries, we can use it directly.
for idx, bs in pairs(unit_basenames) do
basenames[idx] = bs
paths[idx] = unit_paths[idx]
basename_to_index[bs] = idx
end
p = body_end
::continue::
end
return basename_to_index, basenames, paths
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- I/O helpers: atoms source-map + native directory glob -- I/O helpers: atoms source-map + native directory glob
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
+13 -13
View File
@@ -3,16 +3,16 @@
# Wrapper for the tape-atom step-debug helpers. # Wrapper for the tape-atom step-debug helpers.
# The 9 user commands are defined here as STUBS (degraded-state messages). # The 9 user commands are defined here as STUBS (degraded-state messages).
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua` # The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gen/gdb_tape_atoms_runtime.gdb`. # (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gdb_tape_atoms_runtime.gdb`.
# Sourcing that file RE-DEFINES the commands with real implementations. # Sourcing that file RE-DEFINES the commands with real implementations.
# #
# If `build/gen/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map). # If `build/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
# The user just needs to re-run `build_psyq.ps1` to regenerate. # The user just needs to re-run `build_psyq.ps1` to regenerate.
# ── Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ── # ?? Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ??
define tape_atoms define tape_atoms
echo "[gdb_tape_atoms] STUB: runtime file build/gen/gdb_tape_atoms_runtime.gdb not found." echo "[gdb_tape_atoms] STUB: runtime file build/gdb_tape_atoms_runtime.gdb not found."
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file." echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
end end
document tape_atoms document tape_atoms
@@ -21,35 +21,35 @@ document tape_atoms
end end
define break_atom define break_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document break_atom document break_atom
Set a breakpoint at the start of tape atom <name>. STUB state. Set a breakpoint at the start of tape atom <name>. STUB state.
end end
define step_atom define step_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document step_atom document step_atom
Resume execution until the next atom boundary. STUB state. Resume execution until the next atom boundary. STUB state.
end end
define next_atom define next_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document next_atom document next_atom
Alias for step_atom. STUB state. Alias for step_atom. STUB state.
end end
define where_in_atom define where_in_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document where_in_atom document where_in_atom
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state. Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
end end
define stepi_inside_atom define stepi_inside_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document stepi_inside_atom document stepi_inside_atom
One MIPS-instruction step, then where_in_atom. STUB state. One MIPS-instruction step, then where_in_atom. STUB state.
@@ -89,17 +89,17 @@ document wave_ctx
end end
# ── Source the runtime file (re-defines commands with real impls + data). ── # ?? Source the runtime file (re-defines commands with real impls + data). ??
# Try to source from project-root-relative path first (the typical case). # Try to source from project-root-relative path first (the typical case).
# If the user is in a different CWD, the source will fail and stubs remain. # If the user is in a different CWD, the source will fail and stubs remain.
# The runtime file path is computed relative to the ELF's source map convention (build/gen/gdb_tape_atoms_runtime.gdb). # The runtime file path is computed relative to the ELF's source map convention (build/gdb_tape_atoms_runtime.gdb).
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file... echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above. # Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
# The runtime's `define` blocks are intended to overwrite there's no ambiguity to confirm. # The runtime's `define` blocks are intended to overwrite ? there's no ambiguity to confirm.
set confirm off set confirm off
# Source the runtime file (re-defines commands with real impls + data). # Source the runtime file (re-defines commands with real impls + data).
source build/gen/gdb_tape_atoms_runtime.gdb source build/gdb_tape_atoms_runtime.gdb
set confirm on set confirm on
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations). echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
+33 -90
View File
@@ -7,18 +7,11 @@
--- ---
--- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies. --- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
--- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message. --- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
---
--- Writes `<ctx.out_root>/<dir_basename>.errors.h` once per module, with `#error` directives for findings that the C compile surfaces.
--- `passes/report.lua` renders annotations.txt from `corpus.sources_by_dir`, re-validating each source through `M.validate()`.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
-- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale. -- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
-- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module. -- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local write_file = duffle.write_file
local ensure_dir = duffle.ensure_dir
-- The annotation pass reads the source-derived registries from scan_source: -- The annotation pass reads the source-derived registries from scan_source:
-- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks -- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks
@@ -29,11 +22,11 @@ local ensure_dir = duffle.ensure_dir
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- Absolute path to the source file
--- @field text string -- the full source text --- @field text string -- Full source text
--- @field dir string -- the directory containing the source --- @field dir string -- Directory containing the source
--- @field basename string -- filename without extension --- @field basename string -- Filename without extension
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source) --- @field scan table -- Pre-scanned SourceScan payload (from duffle.scan_source)
--- @class PassCtx --- @class PassCtx
--- @field sources SourceFile[] --- @field sources SourceFile[]
@@ -52,28 +45,28 @@ local ensure_dir = duffle.ensure_dir
--- @field warnings table[] --- @field warnings table[]
--- @class AtomAnnotation --- @class AtomAnnotation
--- @field line integer -- source line of the atom_info call --- @field line integer -- Source line of the atom_info call
--- @field macro string -- the macro name (always "atom_info" in the new shape) --- @field macro string -- Macro name (always "atom_info" in the new shape)
--- @field name string -- the atom name --- @field name string -- Atom name
--- @field kind string -- always "info" --- @field kind string -- Always "info"
--- @field binds string|nil -- Binds_X name if any --- @field binds string|nil -- Binds_X name if any
--- @field reads string[] -- R_* names (read targets) --- @field reads string[] -- R_* names (read targets)
--- @field writes string[] -- R_* names (write targets) --- @field writes string[] -- R_* names (write targets)
--- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed) --- @field errors string[]|nil -- Parse-time errors from scan_source (atom_info body malformed)
--- @class DebugSkipMarker -- sub-shape of scan_source.lua's @class DebugSkipMarker --- @class DebugSkipMarker -- Sub-shape of scan_source.lua's @class DebugSkipMarker
--- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive. --- @field marker_kind string -- Exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive.
--- @field marker_line integer --- @field marker_line integer
--- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false) --- @field args string|nil -- Trimmed text inside the parens (nil when has_parens is false)
--- @field has_parens boolean --- @field has_parens boolean
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form) --- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
--- @field pending boolean -- true while awaiting the following declaration --- @field pending boolean -- true while awaiting the following declaration
--- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot --- @field superseded_by_marker_line integer|nil -- Set on a marker that was bumped out of the pending slot
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed --- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
--- @class Finding --- @class Finding
--- @field line integer -- source line (or 0 for pass-level) --- @field line integer -- Source line (or 0 for pass-level)
--- @field msg string -- finding message --- @field msg string -- Finding message
--- @class Findings --- @class Findings
--- @field errors Finding[] --- @field errors Finding[]
@@ -81,14 +74,14 @@ local ensure_dir = duffle.ensure_dir
--- @field info Finding[] --- @field info Finding[]
--- @class PipeCtx --- @class PipeCtx
--- @field atom_index table<string, AtomAnnotation> -- name -> AtomAnnotation (only kind=="atom") --- @field atom_index table<string, AtomAnnotation> -- Name -> AtomAnnotation (only kind=="atom")
--- @field binds_index table<string, BindsStruct> -- name -> BindsStruct --- @field binds_index table<string, BindsStruct> -- Name -> BindsStruct
--- @field annot_counts table<string, integer> -- name -> annotation count (for unique_annotation check) --- @field annot_counts table<string, integer> -- Name -> annotation count (for unique_annotation check)
--- @field types table<string, RegTypeDefault> -- from scan_source --- @field types table<string, RegTypeDefault> -- From scan_source
--- @field atom_views table<string, AtomViewEntry> -- from scan_source --- @field atom_views table<string, AtomViewEntry> -- From scan_source
--- @field seen_defaults table<string, integer> -- duplicate atom_dbg_reg_default detection --- @field seen_defaults table<string, integer> -- Duplicate atom_dbg_reg_default detection
--- @field seen_field table<string, integer> -- Binds_* -> count of fields (set/checked by check_binds_no_duplicate_fields) --- @field seen_field table<string, integer> -- Binds_* -> count of fields (set/checked by check_binds_no_duplicate_fields)
--- @field _scan SourceScan -- full scan payload (typed-view sub-calls live here) --- @field _scan SourceScan -- Full scan payload (typed-view sub-calls live here)
--- @class AnnotatedResult --- @class AnnotatedResult
--- @field atoms AtomEntry[] --- @field atoms AtomEntry[]
@@ -102,11 +95,10 @@ local ensure_dir = duffle.ensure_dir
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Per-check functions (the CHECK_RULES table's payload) -- Per-check functions (the CHECK_RULES table's payload)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--
--- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[]. --- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
--- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match. --- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
--- Check: every annotated atom must have a matching MipsAtom_(name) declaration. --- Check: Every annotated atom must have a matching MipsAtom_(name) declaration.
--- @param a AtomAnnotation --- @param a AtomAnnotation
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
@@ -119,8 +111,8 @@ local function check_atom_decl_exists(a, pipe_ctx, findings)
end end
end end
--- Check: every atom may have AT MOST ONE annotation. --- Check: Every atom may have AT MOST ONE annotation.
--- Post-loop: needs full-corpus `annot_counts` from pipe_ctx. --- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
--- @param findings Findings --- @param findings Findings
local function check_unique_annotation(pipe_ctx, findings) local function check_unique_annotation(pipe_ctx, findings)
@@ -153,7 +145,7 @@ end
--- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift. --- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
--- Three outcomes: missing (error), mismatch (error), match (info). --- Three outcomes: missing (error), mismatch (error), match (info).
--- @param m MacroEntry --- @param m MacroEntry
--- @param wc table<string, integer> -- the shared word-count table (from ctx.shared.word_counts) --- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
--- @param findings Findings --- @param findings Findings
local function check_macro_word_drift(m, wc, findings) local function check_macro_word_drift(m, wc, findings)
local declared = wc[m.name] local declared = wc[m.name]
@@ -311,7 +303,7 @@ local function check_binds_no_duplicate_fields(_src, pipe_ctx, findings)
end end
end end
-- Check: debug-skip markers must satisfy shape + placement constraints. -- Check: Debug-skip markers must satisfy shape + placement constraints.
--- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding. --- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
--- Priority order (first defect wins): --- Priority order (first defect wins):
--- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`) --- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
@@ -322,12 +314,11 @@ end
--- 6. unsupported target_kind -> marker precedes an unrelated declaration --- 6. unsupported target_kind -> marker precedes an unrelated declaration
--- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua. --- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
--- @param marker DebugSkipMarker --- @param marker DebugSkipMarker
--- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot --- @param _pipe_ctx PipeCtx -- Unused; kept for consistency with per_annot // TODO(Ed): Remove?
--- @param findings Findings --- @param findings Findings
local function check_skip_marker(marker, _pipe_ctx, findings) local function check_skip_marker(marker, _pipe_ctx, findings)
local kind = marker.marker_kind local kind = marker.marker_kind
local line = marker.marker_line local line = marker.marker_line
-- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch. -- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
if marker.has_parens then if marker.has_parens then
@@ -378,8 +369,6 @@ local function check_skip_marker(marker, _pipe_ctx, findings)
end end
--- Warn when a source references an unregistered alias. --- Warn when a source references an unregistered alias.
---
--- R_TapePtr, R_AtomJmp, R_PrimCursor, R_FaceCursor, R_VertBase, and R_OtBase opt in through `#define atom_reg` in lottes_tape.h.
--- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration. --- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
--- @param _src SourceFile --- @param _src SourceFile
--- @param pipe_ctx PipeCtx --- @param pipe_ctx PipeCtx
@@ -433,8 +422,7 @@ local CHECK_RULES = {
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Validation -- Validation
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- -- Pure check: Read from src.scan, run validations, emit findings. The scan was done once upstream.
-- Pure check: read from src.scan, run validations, emit findings. The scan was done once upstream.
--- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`. --- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
--- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate. --- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
@@ -480,7 +468,7 @@ end
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx. --- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param src SourceFile --- @param src SourceFile
--- @param corpus_pipe_ctx PipeCtx|nil -- built once per pass from corpus registries; nil builds the same projection here. --- @param corpus_pipe_ctx PipeCtx|nil -- Built once per pass from corpus registries; nil builds the same projection here.
--- @return AnnotatedResult --- @return AnnotatedResult
local function validate(ctx, src, corpus_pipe_ctx) local function validate(ctx, src, corpus_pipe_ctx)
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx) corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
@@ -510,14 +498,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
end end
-- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`. -- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
local seen_defaults = {} local seen_defaults = {}; for reg, _ in pairs (scan.types or {}) do seen_defaults[reg] = (seen_defaults[reg] or 0) + 1 end
for reg, _ in pairs(scan.types or {}) do local atom_infos_list = {}; for _, ai in ipairs(scan.atom_infos or {}) do atom_infos_list[#atom_infos_list + 1] = ai end
seen_defaults[reg] = (seen_defaults[reg] or 0) + 1
end
local atom_infos_list = {}
for _, ai in ipairs(scan.atom_infos or {}) do
atom_infos_list[#atom_infos_list + 1] = ai
end
local pipe_ctx = { local pipe_ctx = {
atom_index = {}, atom_index = {},
@@ -605,40 +587,6 @@ local function validate(ctx, src, corpus_pipe_ctx)
} }
end end
-- ════════════════════════════════════════════════════════════════════════════
-- Per-DIRECTORY (per-module) output: errors.h + annotations.txt
-- ════════════════════════════════════════════════════════════════════════════
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
--- Empty directories (no errors, no atoms) produce no file.
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
if atoms_count == 0 and #errors == 0 then
return nil
end
local out_path = ctx.out_root .. "/" .. dir_basename .. ".errors.h"
local lines = {
"// Auto-generated by ps1_meta.lua (passes/annotation.lua) — DO NOT EDIT",
string.format("// Module: %s Sources: %d", dir_basename, #sources),
"#pragma once",
"",
}
if #errors == 0 then
lines[#lines + 1] = "// annotation pass OK"
else
for _, e in ipairs(errors) do
local src_tag = ""
if e.source then
local src_name = e.source:match("([^/\\]+)$") or e.source
src_tag = src_name .. ": "
end
lines[#lines + 1] = string.format('#error "%s%s (line %d)"', src_tag, e.msg, e.line)
end
end
ensure_dir(ctx.out_root)
write_file(out_path, table.concat(lines, "\n") .. "\n")
return out_path
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M.run — orchestrator entry -- M.run — orchestrator entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -683,11 +631,6 @@ function M.run(ctx)
warnings [#warnings + 1] = { line = w.line, msg = w.msg } warnings [#warnings + 1] = { line = w.line, msg = w.msg }
end end
end end
local err_path = emit_module_errors_h(ctx, dir_basename, dir_atoms, dir_errors, dir_sources)
if err_path then
table.insert(outputs, { errors_h = err_path })
end
end end
return { outputs = outputs, errors = errors, warnings = warnings } return { outputs = outputs, errors = errors, warnings = warnings }
+115 -82
View File
@@ -1,23 +1,21 @@
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms. --- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
--- ---
--- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers: --- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime --- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and
--- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`). --- the gdb-runtime wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
--- ---
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs: one --- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
--- `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes. --- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
--- ---
--- **Two output forms** (per the workspace's per-emission-form pattern from `guide_metaprogram_ssdl.md`): --- Two output forms:
--- 1. **Sourcemap.txt form** — `<out_root>/<basename>.atoms.sourcemap.txt`. Format-version-tagged for forward-compat. --- 1. Markdown form: Handled by `passes/report.lua` (writes `<module>.atoms.md`).
--- Lives in `<out_root>/` (build/gen). Mirrors the convention used by `annotation.lua` --- The render functions `render_source_map` + `render_provenance` are exported for `report.lua` to call directly.
--- (`<out_root>/<basename>.errors.h`) and `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`).
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`. --- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
--- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`. A pure gdb command script — addresses come --- 2. `gdb_tape_atoms_runtime.gdb`: Post-link opt-in (`ctx.flags.gdb_runtime`),
--- from `nm`, the 9 user commands are static `define ... end` blocks. Emitted when `ctx.flags.gdb_runtime` is true --- so the gdb wrapper script + the generated runtime script share the same canonical location.
--- AND `ctx.flags.elf_path` points to an existing ELF. Useful for `gdb-multiarch --without-python` users --- Triggered by `--post-link` or `--gdb-runtime`.
--- (the common case on Windows MinGW builds) — `source <path>` loads it with no Python / Tcl / Guile required.
--- ---
--- **Output format** (sourcemap.txt form): --- Output forma (sourcemap.txt form):
--- ``` --- ```
--- # FORMAT_VERSION 1 --- # FORMAT_VERSION 1
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT --- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
@@ -30,10 +28,7 @@
--- ... --- ...
--- ENDATOM --- ENDATOM
--- ``` --- ```
---
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view. --- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
---
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -43,8 +38,8 @@
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")` -- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
-- at the bottom, so the dofile value IS the duffle module. -- at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local elf_dwarf = require("elf_dwarf") local elf_dwarf = require("elf_dwarf")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -73,8 +68,8 @@ local FORMAT_VERSION = 1
--- @param atom table --- @param atom table
--- @return table[], integer --- @return table[], integer
local function canonical_word_entries(atom) local function canonical_word_entries(atom)
local paths = atom.paths or {} local paths = atom.paths or {}
local events = paths.word_events or {} local events = paths.word_events or {}
local word_items = {} local word_items = {}
for _, item in ipairs(paths.items or {}) do for _, item in ipairs(paths.items or {}) do
if item.kind == "word" then word_items[#word_items + 1] = item end if item.kind == "word" then word_items[#word_items + 1] = item end
@@ -99,41 +94,38 @@ end
--- Render one atom's provenance stanza. Format 1 line shapes: --- Render one atom's provenance stanza. Format 1 line shapes:
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation) --- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
--- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component) --- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
--- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was --- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was declared in `corpus.word_counts`
--- declared in `corpus.word_counts` (populated by word_count_eval + components passes). --- (populated by word_count_eval + components passes).
--- @param src table --- @param src table
--- @param atom table --- @param atom table
--- @param wc table -- identity alias of corpus.word_counts --- @param wc table -- identity alias of corpus.word_counts
--- @return string[], integer --- @return string[], integer
local function emit_provenance_stanza(src, atom, wc) local function emit_provenance_stanza(src, atom, wc)
local lines = {} local lines = {}
local rel_path = src.path:gsub("\\\\", "/") local rel_path = src.path:gsub("\\\\", "/")
local entries, total = canonical_word_entries(atom) local entries, total = canonical_word_entries(atom)
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path) lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
for _, entry in ipairs(entries) do for _, entry in ipairs(entries) do
local inv = entry.invocation local inv = entry.invocation
local macro_count = inv and wc["mac_" .. inv.component_name] local macro_count = inv and wc["mac_" .. inv.component_name]
if inv and macro_count ~= nil then if inv and macro_count ~= nil then
lines[#lines + 1] = string.format( lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d'
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d', , entry.pos, rel_path, entry.line, inv.component_name
entry.pos, rel_path, entry.line, inv.component_name, , inv.def_path or "", inv.def_line or 0, entry.body_line)
inv.def_path or "", inv.def_line or 0, entry.body_line)
else else
lines[#lines + 1] = string.format( lines[#lines + 1] = string.format("WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
end end
end end
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total)) lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
lines[#lines + 1] = "ENDATOM" lines[#lines + 1] = "ENDATOM"
return lines, total return lines, total
end end
--- Render the full provenance file content for one source. --- Render the full provenance file content for one source.
--- @param src table --- @param src table
--- @param wc table --- @param wc table
--- @return string --- @return string
local function render_provenance(src, wc) local function render_provenance(src, wc)
local lines = {} local lines = {}
@@ -166,8 +158,8 @@ end
--- @param wc table --- @param wc table
--- @return string[], integer --- @return string[], integer
local function emit_atom_stanza(src, atom) local function emit_atom_stanza(src, atom)
local lines = {} local lines = {}
local rel_path = src.path:gsub("\\\\", "/") local rel_path = src.path:gsub("\\\\", "/")
local entries, total = canonical_word_entries(atom) local entries, total = canonical_word_entries(atom)
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path) lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
@@ -183,10 +175,10 @@ end
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's --- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
--- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter. --- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
--- @param src table --- @param src table
--- @param wc table --- @param wc table
--- @return string --- @return string
local function render_source_map(src) local function render_source_map(src)
local lines = {} local lines = {}
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT" lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
@@ -220,16 +212,16 @@ end
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries} --- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries}
local function build_atom_table(ctx) local function build_atom_table(ctx)
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path) local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
local corpus = ctx.shared and ctx.shared.corpus local corpus = ctx.shared and ctx.shared.corpus
local matched = {} local matched = {}
for _, src in ipairs(corpus.source_order or {}) do for _, src in ipairs(corpus.source_order or {}) do
local file_base = src.path:match("([^/\\\\]+)$") or src.path local file_base = src.path:match("([^/\\\\]+)$") or src.path
local function append(atom) local function append(atom)
if not atom.paths then return end if not atom.paths then return end
local name = atom.raw_name or atom.name local name = atom.raw_name or atom.name
local info = addrs[name] local info = addrs[name]
if not info then return end if not info then return end
local entries, total = canonical_word_entries(atom) local entries, total = canonical_word_entries(atom)
matched[#matched + 1] = { matched[#matched + 1] = {
@@ -252,9 +244,10 @@ local function build_atom_table(ctx)
return matched return matched
end end
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`, the convenience --- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`,
--- vars set in `emit_gdb_runtime` provide printf args, and each command is a static sequence of `printf` / `tbreak` / --- the convenience vars set in `emit_gdb_runtime` provide printf args, and
--- `if ... end` blocks. The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job. --- each command is a static sequence of `printf` / `tbreak` / `if ... end` blocks.
--- The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
--- ---
--- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while` --- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
--- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path. --- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
@@ -399,8 +392,7 @@ local function append_gdb_commands(lines, matched)
end end
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded --- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
--- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb --- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb source-time.
--- source-time.
--- @param ctx PassCtx --- @param ctx PassCtx
local function emit_gdb_runtime(ctx) local function emit_gdb_runtime(ctx)
if not (ctx.flags and ctx.flags.gdb_runtime) then return end if not (ctx.flags and ctx.flags.gdb_runtime) then return end
@@ -458,7 +450,22 @@ local function emit_gdb_runtime(ctx)
-- Confirmation line for the source operator. -- Confirmation line for the source operator.
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path' lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb" local out_path
-- Move out of `<out_root>/gdb_tape_atoms_runtime.gdb` to `<out_root>/../gdb_tape_atoms_runtime.gdb` when the conventional `<out_root>` is `<build>/gen`
-- (any equivalent spelling — relative, absolute backslash, absolute forward-slash, trailing-separator variants).
-- This puts the gdb runtime alongside the ELF at `build/` rather than under the report subdir.
local function ends_with_gen_dir(p)
if type(p) ~= "string" then return false end
return p:match("[/\\]gen[/\\]?$") ~= nil or p == "build/gen" or p == "build\\gen"
end
if ends_with_gen_dir(ctx.out_root) then
-- Strip the trailing `/gen` segment, then write the runtime script under `build/`.
-- e.g. "C:/projects/Pikuma/ps1/build/gen" -> "C:/projects/Pikuma/ps1/build".
local parent = ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
out_path = parent .. "/gdb_tape_atoms_runtime.gdb"
else
out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
end
duffle.ensure_dir(duffle.dirname(out_path)) duffle.ensure_dir(duffle.dirname(out_path))
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n") duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched)) -- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
@@ -470,10 +477,62 @@ end
local M = {} local M = {}
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`, emit two files -- Expose the pure render functions so `report.lua` and the focused tests can call them directly without triggering the file-emit path.
--- in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt` M.render_source_map = render_source_map
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation). When `ctx.flags.gdb_runtime` M.render_provenance = render_provenance
--- is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
--- Render ONE atom's sourcemap stanza.
--- @param atom table -- atom record (must have `atom.paths` populated)
--- @return string
function M.render_atom_source_map(atom)
assert(type(atom) == "table", "render_atom_source_map: atom must be a table")
assert(type(atom.paths) == "table", "render_atom_source_map: atom.paths must be a table")
local entries, total = canonical_word_entries(atom)
local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
entry.pos, entry.line, entry.text)
end
lines[#lines + 1] = "ENDATOM"
return table.concat(lines, "\n") .. "\n"
end
--- Render ONE atom's provenance stanza — no per-file format header, no enumeration of other atoms.
---
--- `rel_path` is the source path (forward-slashes) embedded in every `CALL` line.
--- The .md caller (report.lua) is expected to derive this once per `## <source>` heading and pass it down for each atom in that source.
--- @param atom table -- atom record (must have `atom.paths` populated)
--- @param wc table -- identity alias of `corpus.word_counts`
--- @param rel_path string -- source path (forward-slashes) for `CALL` fields
--- @return string
function M.render_atom_provenance(atom, wc, rel_path)
assert(type(atom) == "table", "render_atom_provenance: atom must be a table")
assert(type(atom.paths) == "table", "render_atom_provenance: atom.paths must be a table")
assert(type(rel_path) == "string", "render_atom_provenance: rel_path must be a string")
local entries, total = canonical_word_entries(atom)
local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do
local inv = entry.invocation
local macro_count = inv and wc and wc["mac_" .. inv.component_name]
if inv and macro_count ~= nil then
lines[#lines + 1] = string.format(
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
entry.pos, rel_path, entry.line, inv.component_name,
inv.def_path or "", inv.def_line or 0, entry.body_line)
else
lines[#lines + 1] = string.format(
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
end
end
return table.concat(lines, "\n") .. "\n"
end
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
@@ -495,34 +554,8 @@ function M.run(ctx)
} }
end end
-- Always emit the text form (per-source). -- atoms.sourcemap.txt + atoms.provenance.txt content moved to report.lua via `<module>.atoms.md` markdown file.
for _, src in ipairs(corpus.source_order) do -- This pass emits only the post-link gdb_runtime artifact (see emit_gdb_runtime below).
local has_projection = false
for _, atom in ipairs((src.scan or {}).atoms or {}) do
if (atom.kind == "atom" or atom.kind == "raw_atom") and atom.paths then
has_projection = true; break
end
end
if not has_projection then
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
if atom.paths then has_projection = true; break end
end
end
if has_projection then
local basename = duffle.basename_no_ext(src.path)
-- (1) atoms.sourcemap.txt — format-1 per-word call-site map.
local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt"
local sourcemap_body = render_source_map(src)
-- (2) atoms.provenance.txt — format-1 per-word definition/body map.
local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt"
local prov_body = render_provenance(src, wc)
duffle.ensure_dir(duffle.dirname(sourcemap_path))
duffle.write_file_lf(sourcemap_path, sourcemap_body)
duffle.write_file_lf(prov_path, prov_body)
outputs[#outputs + 1] = { kind = "report", path = sourcemap_path }
outputs[#outputs + 1] = { kind = "report", path = prov_path }
end
end
-- Optionally emit the gdb-runtime form (post-link, one file per build). -- Optionally emit the gdb-runtime form (post-link, one file per build).
if ctx.flags and ctx.flags.gdb_runtime then if ctx.flags and ctx.flags.gdb_runtime then
+355
View File
@@ -0,0 +1,355 @@
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
---
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
---
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
--- These GPRs are unavailable to EVERY atom's source pool.
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will
--- exclude R_T4 from that atom's pool.
---
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
--- emit `phase_register_clash` as an info finding (no build stop).
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
---
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
--- emit `phase_register_pool_exhausted` as a build-stopping error.
--- @class AutoRegResult
--- @field outputs table[] -- {kind=, path=} entries
--- @field errors table[] -- {line=, msg=} entries (build-stops)
--- @field warnings table[] -- {line=, msg=} entries (build-continues)
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
--- ════════════════════════════════════════════════════════════════════════════
--- THE GPR ALLOCATION POOL — what is allocatable, and (more importantly) WHY
--- ════════════════════════════════════════════════════════════════════════════
---
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
--- It allocates from a FIXED 10-register pool.
--- This comment block makes the inclusion AND exclusion criteria obvious so a reader doesn't have
--- to grep lottes_tape.h + mips.h to understand the design.
---
--- ── WHAT'S IN THE POOL (10 GPRs, all caller-trash per the O32 ABI) ────────
--- R_T0..R_T7 (GPR codes 8..15), R_V0..R_V1 (GPR codes 2..3)
--- The workhorse of every atom body. The uesr should be aware of atom allocation across atoms they chain.
--- If they have a collision it means either they didn't saturate the register file optimally for a phase,
--- or the may have made the workload to large for the run.
---
--- ── WHAT'S NOT IN THE POOL — and WHY (the "obvious exclusions") ────────────
--- R_T9 (GPR code 25) — R_TapePtr, the tape instruction stream pointer.
--- Owned by the tape runtime (in tape_run / tape_run_a02_s07).
--- `rgcc(R_TapePtr)` register-variable ties the C compiler's view to $t9 across the whole tape_run.
--- The auto-reg pass MUST NOT clobber this; doing so would desync the C-side tape pointer from the
--- hardware pointer and crash on the next tape_run.
---
--- R_T8 (GPR code 24) — R_AtomJmp, the atom-jump register used by the 4-word yield handshake.
--- Every `mac_yield()` / `mac_yield_tail` does `load_word R_AtomJmp, R_TapePtr, 0` then
--- `jump_reg R_AtomJmp`. The auto-reg pass MUST NOT clobber this either, or the atom dispatcher breaks.
--- Owned by the tape runtime, same family as R_TapePtr.
---
--- R_AT (GPR code 1) — Assembler temporary. Reserved by the MIPS O32 ABI for pseudoinstruction expansion
--- (lottes_tape.h:86, mips.h:93). The ISA's psuedo instructions use it as a scratch temporary.
---
--- R_A0..A3 (codes 4..7) — Function arguments. Used in tape_run_a02_s07, see below.
--- R_S0..S7 (codes 16..23) — Callee-saved. Preserved across C-ABI calls by convention.
--- The `tape_run_a02_s07` variant clobbers them deliberately, but the default `tape_run` does NOT.
--- Kept out of POOL to preserve the conservative default.
--- Add them in a separate "big clobber" pool if/when needed.
---
--- R_K0/K1 (codes 26..27) — Kernel / interrupt handler reserves. Never touched by user code; OS-internal.
--- R_GP/SP/FP/RA (codes 28..31) — Stack frame + return-address. Owned by the C compiler; never allocatable.
--- R_0 (code 0) — Hardwired zero. Cannot be written.
---
local POOL = {
"R_T0", "R_T1", "R_T2", "R_T3",
"R_T4", "R_T5", "R_T6", "R_T7",
"R_V0", "R_V1",
}
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
-- Only the POOL entries matter for auto_reg — non-pool aliases
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
local INT_CODE_TO_POOL_GPR = {
[2] = "R_V0", [3] = "R_V1",
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
}
-- Stable sort for deterministic allocation order.
local function stable_sort_keys(tbl)
local keys = {}
for k in pairs(tbl) do keys[#keys + 1] = k end
table.sort(keys)
return keys
end
-- Allocate one phase's auto-reg mappings.
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
local function allocate_phase(phase_label, decls)
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
local pool = {}
for i = 1, #POOL do pool[i] = POOL[i] end
local result = {}
local errors = {}
for _, sym in ipairs(stable_sort_keys(decls)) do
local next_gpr = table.remove(pool, 1)
if not next_gpr then
errors[#errors + 1] = {
line = 0,
msg = string.format("phase_register_pool_exhausted: "
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
.. "(max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
, phase_label, sym),
}
return result, errors
end
result[sym] = next_gpr
end
return result, errors
end
-- Build two projections from corpus.register_alias_registry:
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
local function build_user_pins(corpus)
local user_pinned = {}
local alias_to_gpr = {}
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do
if alias_entry.has_atom_reg and alias_entry.code then
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code]
if gpr then
user_pinned[gpr] = true
alias_to_gpr[alias_name] = gpr
end
end
end
return user_pinned, alias_to_gpr
end
-- Find every physical GPR referenced in the atom body, via EITHER:
-- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
-- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
-- Returns { [physical_gpr_ident] = count }. Clash-detection and source-pool-exclusion logic
-- only needs the presence of each GPR (boolean test), but keeping count preserves the
-- original find_hardcoded_rn shape so callers can switch without churn.
-- The alias pattern is sorted lexicographically to keep the regex deterministic.
local function find_used_gprs(body_text, alias_to_gpr)
local found = {}
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do
found[gpr] = (found[gpr] or 0) + 1
end
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
-- Sorted by name so the regex is byte-stable across runs.
if alias_to_gpr and next(alias_to_gpr) then
local aliases = {}
for alias_name in pairs(alias_to_gpr) do
aliases[#aliases + 1] = alias_name
end
table.sort(aliases)
local pattern = "(" .. table.concat(aliases, "|") .. ")"
for alias_name in body_text:gmatch(pattern) do
local gpr = alias_to_gpr[alias_name]
if gpr and not found[gpr] then
found[gpr] = 1
end
end
end
return found
end
-- Emit one gen/auto_reg.h header per directory.
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
if not mappings or next(mappings) == nil then return end
local out_path = out_dir .. "/" .. "auto_reg.h"
duffle.ensure_dir(out_dir)
local lines = {
"#ifdef INTELLISENSE_DIRECTIVES",
"#pragma once",
"#endif",
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
"// Directory: " .. dir:gsub("/", "\\"),
}
for _, src in ipairs(sources) do
lines[#lines + 1] = "// source: " .. src.path
end
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
lines[#lines + 1] = ""
for _, sym in ipairs(stable_sort_keys(mappings)) do
local gpr = mappings[sym]
local gpr_code = gpr .. "_Code"
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
end
lines[#lines + 1] = ""
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
print(" -> " .. out_path)
return out_path
end
-- ════════════════════════════════════════════════════════════════════════════
-- Pass entry
-- ════════════════════════════════════════════════════════════════════════════
local M = {}
--- @param ctx PassCtx
--- @return AutoRegResult
function M.run(ctx)
local outputs = {}
local errors = {}
local warnings = {}
local corpus = ctx.shared and ctx.shared.corpus
if type(corpus) ~= "table" then
error("auto_reg.run requires ctx.shared.corpus", 0)
end
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
local user_pinned, alias_to_gpr = build_user_pins(corpus)
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
local phase_allocations = {}
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do
local mapping, errs = allocate_phase(phase_label, decls)
for sym, gpr in pairs(mapping) do
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
phase_allocations[phase_label][sym] = gpr
end
for _, e in ipairs(errs) do
errors[#errors + 1] = e
end
end
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
-- Otherwise, allocate a private pool for the atom.
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
local atom_name_to_phase = {}
for phase_label, entry in pairs(corpus.atom_phases or {}) do
for _, atom_name in ipairs(entry.atoms or {}) do
atom_name_to_phase[atom_name] = phase_label
end
end
local atom_allocations = {}
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do
local phase_label = atom_name_to_phase[atom_scope]
-- Build the atom's source pool: start with the full POOL, subtract:
-- (a) every GPR already committed (phase allocations + prior atom allocations)
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
local used = {}
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
-- Folded into `used` so the source_pool exclusion is a single check.
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
if atom and atom.body then
local body_used = find_used_gprs(atom.body, alias_to_gpr)
for gpr in pairs(body_used) do used[gpr] = true end
end
local source_pool = {}
for _, gpr in ipairs(POOL) do
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers
-- declared via atom_reg + _Code defs, preserved across atoms globally).
if not used[gpr] and not user_pinned[gpr] then
source_pool[#source_pool + 1] = gpr
end
end
local result = {}
for _, sym in ipairs(stable_sort_keys(decls)) do
local next_gpr = table.remove(source_pool, 1)
if not next_gpr then
errors[#errors + 1] = {
line = 0,
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
.. "but no free registers remain in its scope pool."
, atom_scope, sym),
}
else
result[sym] = next_gpr
end
end
atom_allocations[atom_scope] = result
end
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
-- This warning is kept as a defensive safety net for cases the body scanner might miss
-- (e.g. macros that expand to register references the scanner cannot resolve).
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
for atom_scope, decls in pairs(atom_allocations) do
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
if atom and atom.body then
local used_in_body = find_used_gprs(atom.body, alias_to_gpr)
for sym, allocated_gpr in pairs(decls) do
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
warnings[#warnings + 1] = {
line = atom.line or 0,
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
, atom_scope, allocated_gpr, sym, allocated_gpr),
}
end
end
end
end
-- 4. Emit per-directory gen/auto_reg.h.
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
local sources_by_dir = corpus.sources_by_dir or {}
for dir, sources in pairs(sources_by_dir) do
local per_dir_mappings = {}
for _, src in ipairs(sources) do
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do
per_dir_mappings[sym] = gpr
end
end
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do
per_dir_mappings[sym] = gpr
end
end
end
local out_dir = dir .. "/gen"
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings)
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
end
return { outputs = outputs, errors = errors, warnings = warnings }
end
return M
+250 -87
View File
@@ -3,13 +3,15 @@
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`. --- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward. --- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
--- ---
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations, --- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations (kind="comp_bare" / "comp_proc"),
--- then resolves the function-args string from the preceding `FI_ MipsAtom ac_X(...)` declaration via a backward walk. --- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
--- ---
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation. --- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, --- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- Lua 5.3 compatible. --- All sources inside the same directory contribute to the same file (per-directory aggregation).
--- The directory itself is the namespace, so the filename does not repeat the module name.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -29,7 +31,7 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Atom component declaration identifiers. -- Atom component declaration identifiers.
local ATOM_COMP_PROC = "MipsAtomComp_Proc_" local ATOM_COMP_PROC = "MipsAtomComp_Proc_"
local MIPS_ATOM = "MipsAtom" -- prefix on the function declaration that wraps an AtomComp_Proc_ local MIPS_ATOM = "Slice_MipsCode" -- prefix on the function declaration that wraps an AtomComp_Proc_
-- Component-name prefixes. -- Component-name prefixes.
local AC_PREFIX = "ac_" -- arg to MipsAtomComp_(ac_X); the X is the atom name local AC_PREFIX = "ac_" -- arg to MipsAtomComp_(ac_X); the X is the atom name
@@ -41,43 +43,44 @@ local MAC_PREFIX_LEN = 4
local BYTE_NEWLINE = 10 local BYTE_NEWLINE = 10
local BYTE_SLASH = 47 local BYTE_SLASH = 47
-- Source dir basename used as the output `.macs.h` filename. -- Output gen subdirectory + filename (per-directory aggregation; the directory name is the namespace).
local GEN_SUBDIR = "gen" local GEN_SUBDIR = "gen"
local MACS_FILENAME = "macs.h"
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Type declarations -- Type declarations
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- Absolute path to the source file
--- @field text string -- the full source text --- @field text string -- Full source text
--- @field dir string -- the directory containing the source --- @field dir string -- Directory containing the source
--- @field basename string -- filename without extension --- @field basename string -- Filename without extension
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source) --- @field scan table -- Pre-scanned SourceScan payload (from duffle.scan_source)
--- @class PassCtx --- @class PassCtx
--- @field sources SourceFile[] -- all source files in the build --- @field sources SourceFile[] -- All source files in the build
--- @field metadata_path string -- path to word_count.metadata.h --- @field metadata_path string -- Path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- Cross-pass shared state
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- Output root (e.g. "build/gen")
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- Project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- Per-pass upstream outputs
--- @field flags table -- CLI flags --- @field flags table -- CLI flags
--- @field verbose boolean -- log diagnostic info --- @field verbose boolean -- Log diagnostic info
--- @class PassResult --- @class PassResult
--- @field outputs table[] -- {kind=, path=} entries describing emit files --- @field outputs table[] -- {kind=, path=} entries describing emit files
--- @field errors table[] -- {line=, msg=} entries; build-stops --- @field errors table[] -- {line=, msg=} entries; build-stops
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds --- @field warnings table[] -- {line=, msg=} entries; build-succeeds
--- @class Component --- @class Component
--- @field name string -- atom name (without `ac_` prefix) --- @field name string -- Atom name (without `ac_` prefix)
--- @field body string -- brace-delimited body (without the braces) --- @field body string -- Brace-delimited body (without the braces)
--- @field args string|nil -- function-args string (function form only) --- @field args string|nil -- Function-args string (function form only)
--- @field line integer -- source line of the declaration --- @field line integer -- Source line of the declaration
--- @field comment string|nil -- scanner-owned `declaration_comment`; the components pass reads it from the scanner record --- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
--- @field kind string -- "comp_bare" | "comp_proc" --- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component — see `project_components`)
--- @field debug_skip boolean -- mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration --- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Local helpers (file I/O + path normalization) -- Local helpers (file I/O + path normalization)
@@ -97,9 +100,9 @@ local M = {}
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found. --- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
--- ---
--- Convention: function form is --- Convention: function form is
--- `FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })` --- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens. --- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
--- We then verify the preceding context ends with `MipsAtom` --- We then verify the preceding context ends with `Slice_MipsCode`
--- (the function-decl keyword with possible qualifiers between). --- (the function-decl keyword with possible qualifiers between).
--- ---
--- @param source string --- @param source string
@@ -200,7 +203,16 @@ end
local function project_components(source, scan) local function project_components(source, scan)
local out = {} local out = {}
for _, a in ipairs(scan.atoms) do for _, a in ipairs(scan.atoms) do
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
if a.kind == "comp_bare" or a.kind == "comp_proc" then if a.kind == "comp_bare" or a.kind == "comp_proc" then
-- Function-args lookup is meaningful for `MipsAtomComp_Proc_` components
-- (the macro sits inside `FI_ Slice_MipsCode ac_X(...)`); the alias expansion
-- discards the `ab` (atom-builder) arg the same way both forms do.
local args = find_function_args_for(source, a.raw_name, a.ident_pos) local args = find_function_args_for(source, a.raw_name, a.ident_pos)
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker. -- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
-- The pass reads `declaration_comment` directly. -- The pass reads `declaration_comment` directly.
@@ -231,7 +243,6 @@ end
-- --
-- Skips `//` sequences that are inside string or character literals -- Skips `//` sequences that are inside string or character literals
-- (a rough heuristic — sufficient for component bodies which don't have those constructs). -- (a rough heuristic — sufficient for component bodies which don't have those constructs).
--
--- @param s string --- @param s string
--- @return string --- @return string
local function convert_line_comments_to_block(s) local function convert_line_comments_to_block(s)
@@ -300,7 +311,9 @@ local function word_count_rec(name, comp_by_name, wc, cache)
local trimmed = t.tok local trimmed = t.tok
if trimmed ~= "" then if trimmed ~= "" then
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1)) local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
if lookup and comp_by_name[lookup] then if lookup == "atom_label" or lookup == "atom_offset" then
-- Pure metaprogram anchors; emit zero words.
elseif lookup and comp_by_name[lookup] then
-- It's a `mac_X(...)` call. Recurse. -- It's a `mac_X(...)` call. Recurse.
n = n + word_count_rec(lookup, comp_by_name, wc, cache) n = n + word_count_rec(lookup, comp_by_name, wc, cache)
elseif lookup and wc and wc[lookup] then elseif lookup and wc and wc[lookup] then
@@ -339,6 +352,115 @@ local function count_all_components(components, wc)
return counts return counts
end end
-- ═══════════════════════════════════════════
-- Per-component metadata derivation (replaces the hardcoded `M.GP0_MACRO_CONTRIB` + `M.INSTRUCTION_LATENCY[mac_*]` tables that previously lived in `duffle.lua`).
--
-- Each `MipsAtomComp_(ac_X) { body }` definition in `code/duffle/lottes_tape.h` is the canonical source.
-- The `mac_X(...)` macros are GENERATED from these definitions by `emit_component_macros_h` for tape-side composition;
-- the metaprogram must NEVER walk the generated variants to derive metadata.
-- Always walk the original `MipsAtomComp_` body via `cc.body_tokens`.
-- ═══════════════════════════════════════════
--- (internal) Recursive cycle-cost derivation. Sum `latency[ident]` per emitted instruction in the component body,
--- recursing through nested `mac_*` calls (so `mac_format_g4_color`'s cost = 4 × `mac_pack_color_word`'s cost).
--- Special rule: `mac_yield`'s cost = 0 (per `lottes_tape.h:125-130` "the runtime cost lands in the next atom's prologue").
--- @param name string -- component bare name (e.g. "yield", "pack_color_word")
--- @param comp_by_name table<string, Component>
--- @param latency table<string, integer>
--- @param cache table<string, integer> -- shared memoization; `-1` sentinel detects cycles
--- @return integer
local function cycle_cost_rec(name, comp_by_name, latency, cache)
if cache[name] ~= nil then return cache[name] end
cache[name] = -1
local cc = comp_by_name[name]
local n
if cc then
if name == "yield" then
-- mac_yield's cost is 0 by convention (the runtime cost lands in the next atom's prologue).
n = 0
else
n = 0
local tokens = cc.body_tokens
for _, t in ipairs(tokens) do
local trimmed = t.tok
if trimmed ~= "" then
local ident = duffle.read_ident(trimmed, 1)
if ident and ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
-- Nested `mac_X(...)` call: recurse.
local nested = ident:sub(MAC_PREFIX_LEN + 1)
n = n + cycle_cost_rec(nested, comp_by_name, latency, cache)
else
-- Leaf instruction or pseudo-macro. Look up in INSTRUCTION_LATENCY; default 1.
n = n + (latency[ident] or 1)
end
end
end
end
else
n = 1
end
cache[name] = n
return n
end
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
--- calls in the component body that target `R_PrimCursor` (these are the RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
--- @param name string
--- @param comp_by_name table<string, Component>
--- @param cache table<string, integer>
--- @return integer
local function gp0_contrib_rec(name, comp_by_name, cache)
if cache[name] ~= nil then return cache[name] end
cache[name] = -1
local cc = comp_by_name[name]
local n
if cc then
n = 0
local tokens = cc.body_tokens
for _, t in ipairs(tokens) do
local trimmed = t.tok
if trimmed ~= "" then
local ident = duffle.read_ident(trimmed, 1)
if ident and ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
-- Nested `mac_X(...)` call: recurse.
local nested = ident:sub(MAC_PREFIX_LEN + 1)
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
if trimmed:find("R_PrimCursor", 1, true) then
n = n + 1
end
end
end
end
else
n = 0
end
cache[name] = n
return n
end
--- Compute `cycle_cost` + `gp0_contrib` for every component in `components` in a single pass.
--- Memoization cache is built ONCE (per source) and shared across both helpers so that
--- a nested `mac_Y` reference inside a `mac_X` body computes its values once.
--- @param components Component[]
--- @param latency table<string, integer>
--- @return table<string, {cycle_cost=integer, gp0_contrib=integer}>
local function compute_components_metadata(components, latency)
local comp_by_name = {}
for _, cc in ipairs(components) do comp_by_name[cc.name] = cc end
local cc_cache = {}
local gc_cache = {}
local out = {}
for _, c in ipairs(components) do
out[c.name] = {
cycle_cost = cycle_cost_rec(c.name, comp_by_name, latency, cc_cache),
gp0_contrib = gp0_contrib_rec(c.name, comp_by_name, gc_cache),
}
end
return out
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Per-component emit logic -- Per-component emit logic
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -348,8 +470,8 @@ end
--- @param s string --- @param s string
--- @return string[] --- @return string[]
local function split_comment_lines(s) local function split_comment_lines(s)
local out = {} local out = {}
local pos = 1 local pos = 1
local s_len = #s local s_len = #s
while pos <= s_len do while pos <= s_len do
local nl = s:find("\n", pos, true) local nl = s:find("\n", pos, true)
@@ -364,12 +486,25 @@ local function split_comment_lines(s)
end end
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form). --- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies don't reference `ab`
--- (the builder is only consumed by the procedural `atombuilder_unroll` line that `MipsAtomComp_Proc_` appends after the body).
--- Inline callers therefore don't need to thread a builder context.
--- @param args_str string|nil --- @param args_str string|nil
--- @return string --- @return string
local function signature_from_args(args_str) local function signature_from_args(args_str)
local arg_names = extract_arg_names(args_str) local arg_names = extract_arg_names(args_str)
if arg_names and #arg_names > 0 then if arg_names and #arg_names > 0 then
return table.concat(arg_names, ", ") -- Drop the leading `ab` (atom-builder) first arg if present.
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
if arg_names[1] == "ab" then
table.remove(arg_names, 1)
end
if #arg_names > 0 then
return table.concat(arg_names, ", ")
end
return "..." -- `ab` was the only arg; fall through to variadic
end end
return "..." return "..."
end end
@@ -411,7 +546,7 @@ local function build_component_lines(c, counts)
-- Marker comment: emitted once for every skipped component. -- Marker comment: emitted once for every skipped component.
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source); -- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment. -- This pass projects `c.debug_skip` and emits the marker as a generated comment.
if c.debug_skip then if c.debug_skip then
lines[#lines + 1] = "/* atom_dbg_skip */" lines[#lines + 1] = "/* atom_dbg_skip */"
end end
@@ -424,9 +559,9 @@ local function build_component_lines(c, counts)
local tokens = duffle.split_top_level_commas(c.body) local tokens = duffle.split_top_level_commas(c.body)
for i = 1, #tokens do tokens[i] = duffle.trim(tokens[i]) end for i = 1, #tokens do tokens[i] = duffle.trim(tokens[i]) end
local sig = signature_from_args(c.args) local sig = signature_from_args(c.args)
-- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components). -- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components).
local n = counts[c.name] local n = counts[c.name]
if n > 0 then if n > 0 then
emit_macro_body(lines, c, sig, tokens) emit_macro_body(lines, c, sig, tokens)
@@ -445,9 +580,15 @@ end
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block, --- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition). --- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
--- @param src SourceFile --- @param dir string -- Absolute source directory
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
--- @return string[] --- @return string[]
local function header_boilerplate(src) local function header_boilerplate(dir, sources)
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" }
for _, src in ipairs(sources) do
source_lines[#source_lines + 1] = "// source: " .. duffle.to_absolute_path(src.path)
end
local source_blob = table.concat(source_lines, "\n")
return { return {
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h. -- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
-- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling. -- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling.
@@ -455,7 +596,7 @@ local function header_boilerplate(src)
"#pragma once", "#pragma once",
"#endif", "#endif",
"// Auto-generated by ps1_meta.lua — DO NOT EDIT", "// Auto-generated by ps1_meta.lua — DO NOT EDIT",
"// Source: " .. duffle.to_absolute_path(src.path), source_blob,
"// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)", "// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)",
"", "",
-- Self-contained: define WORD_COUNT if not already defined. -- Self-contained: define WORD_COUNT if not already defined.
@@ -468,30 +609,30 @@ local function header_boilerplate(src)
} }
end end
--- Compute the output path for one source's `.macs.h` file. --- Compute the per-directory output path for `.macs.h`.
--- The pre-rework convention uses the *directory* basename (not the source file basename) --- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
--- e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`. --- The directory name is the namespace; the filename does not repeat it.
--- This matches what the C codebase #includes. --- @param dir string -- Absolute source directory
--- @param src SourceFile --- @return string -- Output directory
--- @return string -- the output directory --- @return string -- Full output path
--- @return string -- the full output path local function compute_macs_h_path(dir)
local function compute_macs_h_path(src) local out_dir = dir .. "/" .. GEN_SUBDIR
local out_dir = src.dir .. "/" .. GEN_SUBDIR local out_path = out_dir .. "/" .. MACS_FILENAME
local out_path = out_dir .. "/" .. duffle.basename_no_ext(src.dir) .. ".macs.h"
return out_dir, out_path return out_dir, out_path
end end
--- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries. --- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff). --- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param src SourceFile --- @param dir string -- Absolute source directory
--- @param components Component[] --- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components) --- @param components Component[] -- Aggregated components from all sources in this directory
--- @return string|nil -- path to the written file (nil if no components) --- @param counts table<string, integer> -- Precomputed word counts (from count_all_components)
local function emit_component_macros_h(ctx, src, components, counts) --- @return string|nil -- Path to the written file (nil if no components)
local function emit_component_macros_h(ctx, dir, sources, components, counts)
if #components == 0 then return nil end if #components == 0 then return nil end
local out_dir, out_path = compute_macs_h_path(src) local out_dir, out_path = compute_macs_h_path(dir)
local lines = header_boilerplate(src) local lines = header_boilerplate(dir, sources)
for _, c in ipairs(components) do for _, c in ipairs(components) do
for _, l in ipairs(build_component_lines(c, counts)) do for _, l in ipairs(build_component_lines(c, counts)) do
@@ -526,40 +667,44 @@ local function update_canonical_word_counts(corpus, components, counts)
end end
--- @class ComponentDef --- @class ComponentDef
--- @field name string -- bare name (without ac_/mac_ prefix) --- @field name string -- Bare name (without ac_/mac_ prefix)
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`) --- @field line integer -- Definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- absolute source path of the definition --- @field path string -- Absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc" --- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component)
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly --- @field debug_skip boolean -- Mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- (internal) Populate `corpus.components` with this source's components-by-name map. --- (internal) Populate `corpus.components` with this source's components-by-name map.
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component"). --- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
--- The pass does NOT write to `ctx.shared.components` (ownership follows the canonical contract). --- The pass does NOT write to `ctx.shared.components`.
--- The `debug_skip` field mirrors the scanner-owned declaration record (`c.debug_skip`).
--- No parallel skip map is built here; consumers that need the per-component skip state read `corpus.components[name].debug_skip` directly. --- No parallel skip map is built here; consumers that need the per-component skip state read `corpus.components[name].debug_skip` directly.
--- @param corpus table -- the corpus --- The `cycle_cost` + `gp0_contrib` fields are populated from `metadata[c.name]` (computed by `compute_components_metadata` against the original `MipsAtomComp_` body).
--- @param corpus table -- the corpus
--- @param src SourceFile --- @param src SourceFile
--- @param components Component[] --- @param components Component[]
local function update_canonical_components(corpus, src, components) --- @param metadata table<string, {cycle_cost=integer, gp0_contrib=integer}>
local function update_canonical_components(corpus, src, components, metadata)
local rel_path = src.path:gsub("\\", "/") local rel_path = src.path:gsub("\\", "/")
for _, c in ipairs(components) do for _, c in ipairs(components) do
-- Keyed by bare name (e.g. `yield`, `load_tri_indices`). -- Keyed by bare name (e.g. `yield`, `load_tri_indices`).
-- The atoms_source_map pass looks up components by bare name from the corpus; -- The atoms_source_map pass looks up components by bare name from the corpus;
-- `mac_` prefix lives at the call-site identifier and is stripped before lookup. -- `mac_` prefix lives at the call-site identifier and is stripped before lookup.
local m = metadata and metadata[c.name] or nil
if corpus.components[c.name] == nil then if corpus.components[c.name] == nil then
corpus.components[c.name] = { corpus.components[c.name] = {
name = c.name, name = c.name,
line = c.line, line = c.line,
path = rel_path, path = rel_path,
kind = c.kind or "comp_bare", kind = c.kind or "comp_bare",
debug_skip = c.debug_skip == true, debug_skip = c.debug_skip == true,
cycle_cost = m and m.cycle_cost or nil,
gp0_contrib = m and m.gp0_contrib or nil,
} }
else else
-- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it. -- A second declaration of the same bare name: record a typed collision so static-analysis + the report can surface it.
-- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record. -- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
local existing = corpus.components[c.name] local existing = corpus.components[c.name]
if existing.path ~= rel_path or existing.line ~= c.line then if existing.path ~= rel_path or existing.line ~= c.line then
local kind = c.kind or "comp_bare" local kind = c.kind or "comp_bare"
local first_kind = existing.kind or "comp_bare" local first_kind = existing.kind or "comp_bare"
corpus.collisions[#corpus.collisions + 1] = { corpus.collisions[#corpus.collisions + 1] = {
kind = "component", kind = "component",
@@ -624,21 +769,39 @@ function M.run(ctx)
-- * `corpus.component_body_index[name]` — body / line_of / source index -- * `corpus.component_body_index[name]` — body / line_of / source index
-- The pass writes to the corpus only; consumers read from the corpus directly. -- The pass writes to the corpus only; consumers read from the corpus directly.
for _, src in ipairs(corpus.source_order) do -- Per-directory aggregation: every source in the same directory contributes to one `gen/macs.h`.
-- project_components reads from src.scan + does backward lookups on src.text -- The directory itself is the namespace. `corpus.sources_by_dir` preserves source-order within each bucket (matches `corpus.source_order`).
local components = project_components(src.text, src.scan) local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order)
if #components > 0 then for dir, sources in pairs(sources_by_dir) do
-- Compute all component word counts once per source. -- Aggregate components from every source in this directory.
-- Use `corpus.word_counts` so the recursive lookup sees both authored-metadata entries -- `project_components` returns nil for sources with no `MipsAtomComp_` declarations; we skip those.
-- (loaded by word_count_eval.run) AND same-source component entries (populated earlier in this loop by `update_canonical_word_counts`). local aggregated_components = {}
local counts = count_all_components(components, corpus.word_counts) local metadata_per_source = {}
local macs_path = emit_component_macros_h(ctx, src, components, counts) for _, src in ipairs(sources) do
local per_source = project_components(src.text, src.scan) or {}
for _, c in ipairs(per_source) do
aggregated_components[#aggregated_components + 1] = c
end
if #per_source > 0 then
metadata_per_source[src] = compute_components_metadata(per_source, duffle.INSTRUCTION_LATENCY)
end
end
if #aggregated_components > 0 then
-- Compute word counts across the aggregated set. `corpus.word_counts` carries the
-- same-source + prior-directory entries so the recursive lookup sees both.
local counts = count_all_components(aggregated_components, corpus.word_counts)
local macs_path = emit_component_macros_h(ctx, dir, sources, aggregated_components, counts)
if macs_path then if macs_path then
outputs[#outputs + 1] = { macs_h = macs_path } outputs[#outputs + 1] = { macs_h = macs_path }
-- Populate the projections AFTER disk emission (so the byte-identical `.macs.h` contract is preserved before any current-count mutation). -- Populate the projections AFTER disk emission (byte-identical `.macs.h` contract).
update_canonical_word_counts(corpus, components, counts) update_canonical_word_counts(corpus, aggregated_components, counts)
update_canonical_components(corpus, src, components) for _, src in ipairs(sources) do
update_canonical_component_body_index(corpus, src, components, src.scan) local per_source = project_components(src.text, src.scan) or {}
if #per_source > 0 then
update_canonical_components(corpus, src, per_source, metadata_per_source[src])
update_canonical_component_body_index(corpus, src, per_source, src.scan)
end
end
end end
end end
end end
+197 -161
View File
@@ -71,9 +71,11 @@ local DW_LNE_set_address = DWARF_LINE_OPS.DW_LNE_set_address
local DW_RLE_end_of_list = DWARF5_RNGLISTS.end_of_list local DW_RLE_end_of_list = DWARF5_RNGLISTS.end_of_list
local DW_RLE_start_length = DWARF5_RNGLISTS.start_length local DW_RLE_start_length = DWARF5_RNGLISTS.start_length
-- File index 11 in the existing main line unit is hello_gte_tape.c. -- File-index lookup for the existing main line unit (Unit 2).
-- The injector extends that unit rather than appending an unreferenced unit. -- Populated at pass start by `init_file_index_lookup(elf_path)` from the runtime ELF (see `elf_dwarf.read_line_unit_file_table`).
local ATOM_SOURCE_FILE_INDEX = 11 local _file_index_by_basename = nil -- [basename] = 1-based line-table file index
local _file_path_by_index = nil -- [1-based index] = full source path (diagnostics / future consumers)
local _default_atom_source_index = nil -- any valid index used in opaque-row fallbacks
-- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31 -- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31
-- (see collect_per_source_registries + by_alias in build_inserted_children). -- (see collect_per_source_registries + by_alias in build_inserted_children).
@@ -98,16 +100,15 @@ local ABBREV_INLINED_SUBROUTINE = 0x6C -- 108: DW_TAG_inlined_subroutine with
-- (each field transitions from tape memory to GPR at load_pc + 8 = MIPS I load-delay slot boundary). -- (each field transitions from tape memory to GPR at load_pc + 8 = MIPS I load-delay slot boundary).
local ABBREV_BIND_VAR_LOCLIST = 0x6D -- 109: DW_TAG_variable no children + DW_AT_type = ref4 + DW_AT_location = sec_offset local ABBREV_BIND_VAR_LOCLIST = 0x6D -- 109: DW_TAG_variable no children + DW_AT_type = ref4 + DW_AT_location = sec_offset
-- Typed-view pointer_type (for the synthetic V4_S2* / V3_S2* / U4* / void* chains). -- Typed-view pointer_type (for the synthetic V4_S2* / V3_S2* / U4* / void* chains).
-- MUST be a fresh abbrev code in the appended table — emitting uleb128(9) collides with GCC's -- MUST be a fresh abbrev code in the appended table — emitting uleb128(9) collides with GCC's existing abbrev 9
-- existing abbrev 9 (a pointer_type that carries DW_AT_byte_size + DW_AT_type), so gdb misparses -- (a pointer_type that carries DW_AT_byte_size + DW_AT_type), so gdb misparses our 4-byte ref4 as (byte_size, type[0..2]) and lands the cursor mid-attribute.
-- our 4-byte ref4 as (byte_size, type[0..2]) and lands the cursor mid-attribute.
local ABBREV_TYPED_VIEW_POINTER = 0x6E -- 110: DW_TAG_pointer_type no children + DW_AT_type = ref4 (typed-view / U4 / void chain) local ABBREV_TYPED_VIEW_POINTER = 0x6E -- 110: DW_TAG_pointer_type no children + DW_AT_type = ref4 (typed-view / U4 / void chain)
-- DWARF5 §7.7.3 loclist opcodes. -- DWARF5 §7.7.3 loclist opcodes.
local DW_LLE_end_of_list = 0x00 local DW_LLE_end_of_list = 0x00
local DW_LLE_start_length = 0x08 local DW_LLE_start_length = 0x08
local DW_OP_reg0 = 0x50 -- base reg op; regN = 0x50 + N local DW_OP_reg0 = 0x50 -- base reg op; regN = 0x50 + N
local DW_OP_breg0 = 0x70 -- base breg op; bregN = 0x70 + N (SLEB offset) local DW_OP_breg0 = 0x70 -- base breg op; bregN = 0x70 + N (SLEB offset)
local DW_OP_piece = 0x93 local DW_OP_piece = 0x93
local MIPS_LOAD_DELAY_BYTES = 0x08 -- 1 load word + 1 BD-slot word local MIPS_LOAD_DELAY_BYTES = 0x08 -- 1 load word + 1 BD-slot word
@@ -153,44 +154,66 @@ local DW_AT_inline = 0x20 -- DWARF5 §7.7.1: DW_AT_inline (used by a
local DW_AT_decl_file = 0x3A -- DWARF5 §7.7.1: DW_AT_decl_file (1-based file index into the CU's file table) local DW_AT_decl_file = 0x3A -- DWARF5 §7.7.1: DW_AT_decl_file (1-based file index into the CU's file table)
local DW_AT_decl_line = 0x3B -- DWARF5 §7.7.1: DW_AT_decl_line local DW_AT_decl_line = 0x3B -- DWARF5 §7.7.1: DW_AT_decl_line
-- File index lookup table for the existing main line unit (Unit 2). -- Replaced the hardcoded `ATOM_SOURCE_FILE_INDEX = 11` and the `PROVENANCE_BASENAME_TO_FILE_INDEX` table below with a runtime lookup
-- Provenance paths come back with mixed slashes; we normalize to basename and look up against the line unit's actual file table. -- (`init_file_index_lookup` + `resolve_provenance_file_index`) that reads the actual `.debug_line` file table from the post-link ELF.
-- Current scope has two provenance basenames: hello_gte_tape.c (the atom's call site) and lottes_tape.h (the component definition).
-- Both live in the existing gcc-generated line unit; their 1-based indices are stable across rebuilds because the include order --- Populate the module-level file-index lookup table from the `.debug_line` section of the post-link ELF pointed at by `elf_path`.
-- in code/gte_hello/hello_gte.c determines the unit's file table. --- This MUST be called exactly once at pass start (from `M.run`) before any `resolve_provenance_file_index` invocation;
-- gcc only adds a file to the line table when it has actual line-number entries; --- downstream callers handle a nil table as "no file info available; fall back to errors".
-- headers that are pure macros/typedefs (dsl.h, memory.h, math.h, mips.h, gp.h, gte.h, etc.) never appear. ---
-- lottes_tape.h is the FIRST include that emits line entries (MipsAtomComp_ declarations), so it is the FIRST entry after the primary file. --- The lookup uses `elf_dwarf.read_line_unit_file_table` (which parses both DWARF3 and DWARF5 line-program units —
local PROVENANCE_BASENAME_TO_FILE_INDEX = { --- the crt0.s assembler-side DWARF5 unit may emit non-standard form codes for paths and is intentionally skipped).
["hello_gte_tape.c"] = ATOM_SOURCE_FILE_INDEX, -- = 11 --- @param elf_path string|nil
["lottes_tape.h"] = 2, local function init_file_index_lookup(elf_path)
} if not elf_path or elf_path == "" then return end
local b2i, _basenames, paths = elf_dwarf.read_line_unit_file_table(elf_path)
if type(b2i) ~= "table" or type(paths) ~= "table" then
io.stderr:write("[dwarf_injection] read_line_unit_file_table returned no file table for: " .. tostring(elf_path) .. "\n")
return
end
_file_index_by_basename = b2i
_file_path_by_index = paths
-- Pick any valid index for the opaque-row fallbacks at lines 466 + 570
-- (both sites legitimately want "any file index"; gdb resolves whatever index we emit to whatever that file's line happens to be).
for idx in pairs(paths) do
_default_atom_source_index = idx
break
end
end
--- Resolve an absolute provenance path to the line-unit file index used by the emitting line program. --- Resolve an absolute provenance path to the line-unit file index used by the emitting line program.
--- Normalizes mixed `/` and `\` separators to a basename and looks it up against the known file table. --- Normalizes mixed `/` and `\` separators to a basename and looks it up against the runtime-computed file table populated by `init_file_index_lookup`.
--- ---
--- Fails loudly on an unknown provenance basename: adding a new component source file requires extending --- Returns 0 (the DWARF `set_file(0)` "no file change" sentinel) when the basename is not in the file table.
--- `PROVENANCE_BASENAME_TO_FILE_INDEX` so the line-program emission contract stays explicit. --- This is a normal occurrence: the compiler only adds a file to the `.debug_line` file table when the file has line-numbered content (i.e., code).
--- Silent fallback to ATOM_SOURCE_FILE_INDEX would mask the new-file case by misattributing component rows to the atom's source file. --- Files containing only static-array data (e.g. `MipsAtomComp_` declarations in `gp.atom.c`, `psyq.atom.c`, `pad.atom.c` — the OT-tag inserts, etc.) produce no line numbers,
--- @param path string -- absolute provenance path (e.g. "C:/.../lottes_thttps://www.youtube.com/watch?v=ORM4yLkdKx8ape.h" or "C:\\...\\lottes_tape.h") --- so gcc omits them from the file table.
--- @return integer -- 1-based line-unit file index --- The DWARF emitter then keeps the previous line-program file state instead of pointing at a file that has no entries to walk.
--- A stderr warning is emitted per-miss so the user can audit which files the compiler dropped.
--- @param path string -- absolute provenance path (mixed slashes accepted)
--- @return integer -- 1-based line-unit file index, or 0 on miss (DWARF no-change sentinel)
local function resolve_provenance_file_index(path) local function resolve_provenance_file_index(path)
if _file_index_by_basename == nil then
error("[dwarf_injection] resolve_provenance_file_index called before init_file_index_lookup. Is M.run being entered correctly (with --elf)?")
end
if path == nil or path == "" then if path == nil or path == "" then
error("[dwarf_injection] resolve_provenance_file_index: empty path") error("[dwarf_injection] resolve_provenance_file_index: empty path")
end end
-- Normalize backslashes → forward slashes (paths arrive with mixed separators from the provenance file: forward slashes from Lua's io.lines; -- Normalize backslashes → forward slashes (paths arrive with mixed separators from the provenance file).
-- backslashes if the input ever round-trips through Windows shell expansion).
local normalized = path:gsub("\\", "/") local normalized = path:gsub("\\", "/")
-- Take the last path component (the basename). -- Take the last path component (the basename).
local basename = normalized:match("([^/]+)$") or normalized local basename = normalized:match("([^/]+)$") or normalized
local idx = PROVENANCE_BASENAME_TO_FILE_INDEX[basename] local idx = _file_index_by_basename[basename]
if idx == nil then if idx ~= nil then return idx end
error(string.format( -- Last-resort exact-path match (handles paths that don't reduce to a known basename).
"[dwarf_injection] resolve_provenance_file_index: unknown provenance basename '%s' (from '%s'). " for i, p in pairs(_file_path_by_index) do
.. "Extend PROVENANCE_BASENAME_TO_FILE_INDEX in passes/dwarf_injection.lua.", if p and p:gsub("\\", "/") == normalized then return i end
basename, path))
end end
return idx -- File is in the corpus but gcc omitted it from the .debug_line file table (data-only content).
-- Return 0 = DWARF `set_file(0)` no-change sentinel so the line program keeps its prior file state.
io.stderr:write(string.format("[dwarf_injection] line-table miss: '%s' (basename '%s') not in .debug_line file table; "
.. "falling back to set_file(0)\n", path, basename))
return 0
end end
local DW_FORM_addr = 0x01 local DW_FORM_addr = 0x01
@@ -205,7 +228,6 @@ local DW_FORM_sec_offset = 0x17 -- 4-byte section-relative offset (into .d
-- DW_OP_reg0 + DW_OP_piece are declared above (lines 114-116) alongside the other DWARF5 §7.7.3 loclist opcodes. -- DW_OP_reg0 + DW_OP_piece are declared above (lines 114-116) alongside the other DWARF5 §7.7.3 loclist opcodes.
local DW_ATE_unsigned = 0x07 -- DWARF5 §7.8.1: DW_ATE_unsigned (used for U4 base type) local DW_ATE_unsigned = 0x07 -- DWARF5 §7.8.1: DW_ATE_unsigned (used for U4 base type)
-- (DW_LANG_Mips_Assembler = 0x8001 was used in the, but we want this CU to look like a C TU so VSCode's Variables pane treats it as code.) -- (DW_LANG_Mips_Assembler = 0x8001 was used in the, but we want this CU to look like a C TU so VSCode's Variables pane treats it as code.)
@@ -425,9 +447,9 @@ end
--- Statement-state rules: --- Statement-state rules:
--- * A marked whole atom emits one opaque is_stmt=false range row and no nested component rows; its subprogram symbol/range remains available. --- * A marked whole atom emits one opaque is_stmt=false range row and no nested component rows; its subprogram symbol/range remains available.
--- * Per-row policy at every other PC: --- * Per-row policy at every other PC:
--- - Call-site row of any invocation's first word: is_stmt = true (unconditional; `want_call = true`). --- - Call-site row of any invocation's first word: is_stmt = true (unconditional; `want_call = true`).
--- - Body row of any invocation (first or subsequent): is_stmt = not inv.debug_skip (`want_body = not inv.debug_skip`). --- - Body row of any invocation (first or subsequent): is_stmt = not inv.debug_skip (`want_body = not inv.debug_skip`).
--- - RAW word (no containing invocation): is_stmt = true (unconditional). --- - RAW word (no containing invocation): is_stmt = true (unconditional).
--- * The previous per-word `marked_idx` ancestor walk and the GDB 12 zero-instruction-prologue duplicate row at atom entry are DELETED; the new --- * The previous per-word `marked_idx` ancestor walk and the GDB 12 zero-instruction-prologue duplicate row at atom entry are DELETED; the new
--- first-word emission IS the entry statement. --- first-word emission IS the entry statement.
--- * Whole-atom suppression wins over component markers; no nested inversion. --- * Whole-atom suppression wins over component markers; no nested inversion.
@@ -461,7 +483,7 @@ local function build_atom_sequence(atom)
if atom.debug_skip then if atom.debug_skip then
return table.concat({ return table.concat({
set_address(atom.addr), set_address(atom.addr),
set_file(ATOM_SOURCE_FILE_INDEX), set_file(resolve_provenance_file_index(atom.src_path)),
advance_line(atom.entries[1].line - 1), advance_line(atom.entries[1].line - 1),
negate_stmt(), negate_stmt(),
copy_op(), copy_op(),
@@ -508,10 +530,9 @@ local function build_atom_sequence(atom)
-- * If the invocation's body has any NESTED invocations (parent_id == top_inv.id), the body's -- * If the invocation's body has any NESTED invocations (parent_id == top_inv.id), the body's
-- first content is the call_line of the earliest nested invocation (by start_pos). -- first content is the call_line of the earliest nested invocation (by start_pos).
-- * Otherwise (only RAW words in the body), it's the line of the first raw word = body_lines[1]. -- * Otherwise (only RAW words in the body), it's the line of the first raw word = body_lines[1].
-- This is the value the multi-row PC's body_lines[1] row must reference for source-order display: -- This is the value the multi-row PC's body_lines[1] row must reference for source-order display: `anc.body_lines[1]` is the line of the FIRST WORD
-- `anc.body_lines[1]` is the line of the FIRST WORD (which for an outer whose body starts with a -- (which for an outer whose body starts with a nested expansion is inside the inner's expansion = wrong for display purposes);
-- nested expansion is inside the inner's expansion = wrong for display purposes); `anc.body_first_line` -- `anc.body_first_line` is the body's first content line in the parent's source (= correct for display).
-- is the body's first content line in the parent's source (= correct for display).
local body_first_line_of = {} local body_first_line_of = {}
for _, top_inv in ipairs(invs) do for _, top_inv in ipairs(invs) do
local earliest_nested_call_line = nil local earliest_nested_call_line = nil
@@ -565,7 +586,7 @@ local function build_atom_sequence(atom)
parts[#parts + 1] = copy_op() parts[#parts + 1] = copy_op()
end end
local call_file_idx = ATOM_SOURCE_FILE_INDEX local call_file_idx = resolve_provenance_file_index(atom.src_path)
-- --- Atom entry (idx 1) ------------------------------------------------- -- --- Atom entry (idx 1) -------------------------------------------------
local entry_1 = atom.entries[1] local entry_1 = atom.entries[1]
@@ -574,13 +595,11 @@ local function build_atom_sequence(atom)
-- If atom entry 1 starts inside an invocation, walk the ancestry and emit a call-site row + (when applicable) -- If atom entry 1 starts inside an invocation, walk the ancestry and emit a call-site row + (when applicable)
-- a body_lines[1] row for every active ancestor. For a non-nested invocation this is just the one pair; -- a body_lines[1] row for every active ancestor. For a non-nested invocation this is just the one pair;
-- for nested invocations this emits the outer call-site + body_lines[1] rows BEFORE the inner pair so the debugger displays -- for nested invocations this emits the outer call-site + body_lines[1] rows BEFORE the inner pair so the debugger displays
-- the outer body line at the inner's first word -- the outer body line at the inner's first word (PROBLEM B fix).
-- (PROBLEM B fix).
-- --
-- A marked OUTERMOST ancestor's body_lines[1] row is suppressed at this PC (the existing full-skip -- A marked OUTERMOST ancestor's body_lines[1] row is suppressed at this PC (the existing full-skip contract is preserved for the marked outer range);
-- contract is preserved for the marked outer range); its call-site row IS still emitted as a -- Its call-site row IS still emitted as a statement.
-- statement. Marked INNER ancestors always emit their body_lines[1] row with is_stmt=false -- Marked INNER ancestors always emit their body_lines[1] row with is_stmt=false (the per-invocation `want_body = not inv.debug_skip` predicate).
-- (the per-invocation `want_body = not inv.debug_skip` predicate).
if #entry_1_ancestry == 0 then if #entry_1_ancestry == 0 then
-- RAW word at atom entry: single call-site row, always a statement target. -- RAW word at atom entry: single call-site row, always a statement target.
emit_row(call_file_idx, entry_1.line, true) emit_row(call_file_idx, entry_1.line, true)
@@ -588,15 +607,12 @@ local function build_atom_sequence(atom)
-- Atom starts in an invocation. Walk the ancestry outermost-first. -- Atom starts in an invocation. Walk the ancestry outermost-first.
-- Each ancestor emits one call-site row (statement) and one body_lines[1] row -- Each ancestor emits one call-site row (statement) and one body_lines[1] row
-- (statement iff unmarked; suppressed for marked outermost). -- (statement iff unmarked; suppressed for marked outermost).
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content -- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
-- line in the parent's source), NOT anc.body_lines[1] (= the line of the first WORD, -- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call).
-- which is wrong when the outer's body starts with a nested call).
for ai, anc in ipairs(entry_1_ancestry) do for ai, anc in ipairs(entry_1_ancestry) do
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model") assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
assert(anc.body_lines[1] ~= nil assert(anc.body_lines[1] ~= nil, "dwarf_injection: body_lines[1] missing on first-word entry for inv=" .. tostring(anc.component_name))
, "dwarf_injection: body_lines[1] missing on first-word entry for inv=" .. tostring(anc.component_name)) assert(anc.call_path and anc.call_path ~= "", "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
assert(anc.call_path and anc.call_path ~= ""
, "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true) emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
local is_outermost = (ai == 1) local is_outermost = (ai == 1)
if not (is_outermost and anc.debug_skip) then if not (is_outermost and anc.debug_skip) then
@@ -615,29 +631,23 @@ local function build_atom_sequence(atom)
if inv and idx == inv.start_pos + 1 then if inv and idx == inv.start_pos + 1 then
-- First word of the innermost active invocation (PROBLEM B fix — nested-display rule). -- First word of the innermost active invocation (PROBLEM B fix — nested-display rule).
-- Walk the active ancestry outermost-first; for each ancestor emit a call-site row -- Walk the active ancestry outermost-first; for each ancestor emit a call-site row (statement) + a body_lines[1] row.
-- (statement) + a body_lines[1] row. The inner-most invocation's call-site + body pair -- The inner-most invocation's call-site + body pair become the LAST two rows in the sequence.
-- become the LAST two rows in the sequence. Marked outermost ancestors suppress their -- Marked outermost ancestors suppress their body_lines[1] row at this PC (the existing full-skip contract is preserved for the marked outer range);
-- body_lines[1] row at this PC (the existing full-skip contract is preserved for the -- all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
-- marked outer range); all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
-- --
-- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC -- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC
-- for debugger context: source-level stepping now shows the outer body line (not the -- for debugger context: source-level stepping now shows the outer body line (not the inner body line) when stepping into the inner. PROBLEM B fix.
-- inner body line) when stepping into the inner. PROBLEM B fix. -- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content -- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call:
-- line in the parent's source), NOT anc.body_lines[1] (= the line of the first WORD, -- gdb 12.1 picks the displayed line as the LAST row at the same PC in byte-stream order,
-- which is wrong when the outer's body starts with a nested call: gdb 12.1 picks the -- so the disc=1 row's value matters for what's shown when stepping into the nested case).
-- displayed line as the LAST row at the same PC in byte-stream order, so the disc=1 row's
-- value matters for what's shown when stepping into the nested case).
local ancestry = ancestry_idx[idx] local ancestry = ancestry_idx[idx]
for ai, anc in ipairs(ancestry) do for ai, anc in ipairs(ancestry) do
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model") assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
assert(anc.body_lines[1] ~= nil assert(anc.body_lines[1] ~= nil, string.format("missing body_lines[1] for inv=%s start_pos=%d len=%d", anc.component_name, anc.start_pos, #(anc.body_lines or {})))
, string.format("missing body_lines[1] for inv=%s start_pos=%d len=%d", assert(anc.call_path and anc.call_path ~= "", "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
anc.component_name, anc.start_pos, #(anc.body_lines or {}))) emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
assert(anc.call_path and anc.call_path ~= ""
, "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
local is_outermost = (ai == 1) local is_outermost = (ai == 1)
if not (is_outermost and anc.debug_skip) then if not (is_outermost and anc.debug_skip) then
emit_row(resolve_provenance_file_index(anc.def_path), body_first_line_of[anc.id] or anc.body_lines[1], not anc.debug_skip) emit_row(resolve_provenance_file_index(anc.def_path), body_first_line_of[anc.id] or anc.body_lines[1], not anc.debug_skip)
@@ -647,16 +657,13 @@ local function build_atom_sequence(atom)
-- Subsequent body word of the innermost active invocation: `body_lines[k]` is indexed by the 1-based offset of this word inside the invocation. -- Subsequent body word of the innermost active invocation: `body_lines[k]` is indexed by the 1-based offset of this word inside the invocation.
-- Both `idx` (1-based DWARF entry index) and `inv.start_pos` (0-based emitted-word position stamped at `emit_invoke_begin`) come from the same -- Both `idx` (1-based DWARF entry index) and `inv.start_pos` (0-based emitted-word position stamped at `emit_invoke_begin`) come from the same
-- monotonic counter, so `idx - inv.start_pos` is exactly the 1-based k (the first word of the invocation has `idx == inv.start_pos + 1`, hence `k == 1`). -- monotonic counter, so `idx - inv.start_pos` is exactly the 1-based k (the first word of the invocation has `idx == inv.start_pos + 1`, hence `k == 1`).
-- atom_dbg_step_ux_20260725: `want_body = not inv.debug_skip`. The previous `want = not marked_idx[idx]` -- atom_dbg_step_ux_20260725: `want_body = not inv.debug_skip`.
-- (which suppressed ALL body rows when any ancestor was marked) is replaced by the per-invocation -- The previous `want = not marked_idx[idx]` (which suppressed ALL body rows when any ancestor was marked) is replaced by the per-invocation predicate.
-- predicate. Marked invocations emit non-statement body rows at every body word; unmarked -- Marked invocations emit non-statement body rows at every body word; unmarked invocations emit statement body rows.
-- invocations emit statement body rows.
assert(inv.body_lines, "missing body_lines: emitter did not run emission-model") assert(inv.body_lines, "missing body_lines: emitter did not run emission-model")
local words_into = idx - inv.start_pos local words_into = idx - inv.start_pos
assert(inv.body_lines[words_into] ~= nil assert(inv.body_lines[words_into] ~= nil, string.format("missing body_lines[%d] for inv=%s start_pos=%d len=%d idx=%d", words_into, inv.component_name, inv.start_pos, #(inv.body_lines or {}), idx))
, string.format("missing body_lines[%d] for inv=%s start_pos=%d len=%d idx=%d", emit_row(resolve_provenance_file_index(inv.def_path), inv.body_lines[words_into], not inv.debug_skip)
words_into, inv.component_name, inv.start_pos, #(inv.body_lines or {}), idx))
emit_row(resolve_provenance_file_index(inv.def_path), inv.body_lines[words_into], not inv.debug_skip)
else else
-- RAW word: single call-site row, always a statement target (the word itself is unmarked). -- RAW word: single call-site row, always a statement target (the word itself is unmarked).
emit_row(call_file_idx, entry.line, true) emit_row(call_file_idx, entry.line, true)
@@ -696,15 +703,17 @@ end
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]` --- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
--- is the k-th word's source line within the component body. --- is the k-th word's source line within the component body.
--- ---
--- @param corpus table -- the corpus from `ctx.shared.corpus` --- @param corpus table -- From `ctx.shared.corpus`
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm` --- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
--- @return table[] -- list of {name, addr, size_bytes, words, entries, invocations, debug_skip?} --- @return table[] -- List of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
local function build_atom_table(corpus, addrs) local function build_atom_table(corpus, addrs)
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr. -- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
local atoms_by_name = corpus.atoms_by_name or {} local atoms_by_name = corpus.atoms_by_name or {}
-- Per-atom ingest. Returns nil if the atom is absent from the corpus; the caller skips it via the `if atom then ...` guard. -- Per-atom ingest. Returns nil if the atom is absent from the corpus; the caller skips it via the `if atom then ...` guard.
local function ingest_atom(name, info) -- `src_path` is the absolute source path that declared this atom; the build_atom_table iteration below threads `src.path` through.
-- This is consumed by `build_atom_sequence::set_file(...)` for opaque-row fallbacks + raw-word rows (atoms where no invocation ancestry exists).
local function ingest_atom(name, info, src_path)
local atom_record = atoms_by_name[name] local atom_record = atoms_by_name[name]
if not atom_record then return nil end if not atom_record then return nil end
@@ -712,8 +721,8 @@ local function build_atom_table(corpus, addrs)
local word_events = paths.word_events or {} local word_events = paths.word_events or {}
local invocations_proj = paths.invocations or {} local invocations_proj = paths.invocations or {}
-- Build the dense entries list from `word_events`. -- Build the dense entries list from `word_events`.
-- `word_events[i].i` = the 0-based `.word` position -- `word_events[i].i` = the 0-based `.word` position
-- `call_line` = the root atom's physical source line for that word (stamped by emission_model) -- `call_line` = the root atom's physical source line for that word (stamped by emission_model)
local entries = {} local entries = {}
for idx, ev in ipairs(word_events) do for idx, ev in ipairs(word_events) do
entries[#entries + 1] = { entries[#entries + 1] = {
@@ -730,6 +739,7 @@ local function build_atom_table(corpus, addrs)
words = #word_events, words = #word_events,
entries = entries, entries = entries,
debug_skip = atom_record.debug_skip == true, debug_skip = atom_record.debug_skip == true,
src_path = src_path or "",
} }
-- Consume invocation records from `atom.paths.invocations`. It is the single producer of per-invocation body_lines, per-invocation debug_skip, -- Consume invocation records from `atom.paths.invocations`. It is the single producer of per-invocation body_lines, per-invocation debug_skip,
@@ -753,9 +763,25 @@ local function build_atom_table(corpus, addrs)
end end
local out = {} local out = {}
for name, info in pairs(addrs) do -- Walk every source's atom list (which preserves source order + per-source src_path).
local atom = ingest_atom(name, info) -- Cross-ref with the nm symbol table; atoms absent from `addrs` are skipped
if atom then out[#out + 1] = atom end -- (an atom declared in source but not emitted as a symbol is a metaprogram or atom-info bug, not a source-correlation bug — emit_no_emit would catch it upstream).
for _, src in ipairs((corpus and corpus.source_order) or {}) do
local src_path = src.path or ""
for _, atom_rec in ipairs(((src.scan or {}).atoms) or {}) do
local info = addrs[atom_rec.name or atom_rec.raw_name]
if info then
local atom = ingest_atom(atom_rec.name or atom_rec.raw_name, info, src_path)
if atom then out[#out + 1] = atom end
end
end
for _, atom_rec in ipairs(((src.scan or {}).raw_atoms) or {}) do
local info = addrs[atom_rec.name or atom_rec.raw_name]
if info then
local atom = ingest_atom(atom_rec.name or atom_rec.raw_name, info, src_path)
if atom then out[#out + 1] = atom end
end
end
end end
table.sort(out, function(a, b) return a.addr < b.addr end) table.sort(out, function(a, b) return a.addr < b.addr end)
return out return out
@@ -800,27 +826,33 @@ end
--- load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)), --- load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
--- ... --- ...
--- ---
--- Also matches `load_half` / `load_half_u` / `load_byte` / `load_byte_u` (any MIPS load instruction with `(R_<reg>, R_<base>, O_(<Binds_X>, FieldName))` shape).
--- Every field's `byte_size` + `offset` determine which load to emit; this function only records the (reg, field) pair.
---
--- The GPR for each `R_<reg>` is looked up in the merged register_alias_registry; aliases absent from the registry --- The GPR for each `R_<reg>` is looked up in the merged register_alias_registry; aliases absent from the registry
--- (no `atom_reg` opt-in) are silently skipped — the resulting rbind record will be incomplete and the atom will fail to bind a usable piece chain. --- (no `atom_reg` opt-in) are silently skipped — the resulting rbind record will be incomplete and the atom will fail to bind a usable piece chain.
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in. --- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
--- ---
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level --- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level statements (each entry is a single `load_*` call or other statement).
--- statements (each entry is a single `load_word(...)` call or other statement). --- @param body_tokens table[] -- The atom's pre-tokenized body statements (from atom.body_tokens)
--- @param body_tokens table[] -- the atom's pre-tokenized body statements (from atom.body_tokens) --- @param binds_name string -- Expected Binds_X name (skip pairs with mismatching binds)
--- @param binds_name string -- expected Binds_X name (skip pairs with mismatching binds) --- @param registries table -- Merged registries from collect_per_source_registries
--- @param registries table -- merged registries from collect_per_source_registries --- @return table[] -- List of {reg = <MIPS index>, field = <field name>}
--- @return table[] -- list of {reg = <MIPS index>, field = <field name>}
local function parse_body_load_pairs(body_tokens, binds_name, registries) local function parse_body_load_pairs(body_tokens, binds_name, registries)
local pairs = {} local pairs = {}
local reg_index_by_name = (registries and registries.register_alias_registry) or {} local reg_index_by_name = (registries and registries.register_alias_registry) or {}
-- One regex that matches any of: load_word, load_half, load_half_u, load_byte, load_byte_u, gte_lw, gte_lwc2.
-- The captured ident is `kind`; `inner` holds the parens body for arg parsing.
local load_pattern = "^(load_word|load_half|load_half_u|load_byte|load_byte_u|gte_lw|gte_lwc2)%s*%((.*)%)$"
for _, t in ipairs(body_tokens or {}) do for _, t in ipairs(body_tokens or {}) do
local tok = duffle.trim(t.tok or "") local tok = duffle.trim(t.tok or "")
-- Match "load_word(...)" — the entire call is one body_tokens entry. local kind, inner = tok:match(load_pattern)
local inner = tok:match("^load_word%s*%((.*)%)$") if kind then
if inner then
local args = duffle.split_top_level_commas(inner) local args = duffle.split_top_level_commas(inner)
-- Expected shape: (R_<reg>, R_TapePtr, O_(Binds_<X>, FieldName)) -- Expected shape for an rbind piece-chain load: (R_<reg>, R_TapePtr, O_(Binds_<X>, FieldName))
if #args >= 3 then -- The second arg MUST be R_TapePtr — loads from other bases (e.g. `load_byte_u(R_RawStatus, R_PadRaw, 0)`)
-- are field-derivative loads that read already-bound tape values; they're NOT a new piece-chain.
if #args >= 3 and duffle.trim(args[2]) == "R_TapePtr" then
local reg_name = duffle.trim(args[1]) local reg_name = duffle.trim(args[1])
local third_arg = duffle.trim(args[3]) local third_arg = duffle.trim(args[3])
-- Match O_(Binds_<X>, FieldName) -- Match O_(Binds_<X>, FieldName)
@@ -839,9 +871,7 @@ local function parse_body_load_pairs(body_tokens, binds_name, registries)
end end
--- Collect every rbind atom + the matching Binds_X struct + (reg, field) pairs. --- Collect every rbind atom + the matching Binds_X struct + (reg, field) pairs.
---
--- Inputs come from the dep-closed `scan-source` pass (the per-source `src.scan` payload is preserved on each `corpus.source_order` entry). --- Inputs come from the dep-closed `scan-source` pass (the per-source `src.scan` payload is preserved on each `corpus.source_order` entry).
---
--- Returns: --- Returns:
--- rbind_atoms = {[atom_name] = {binds, fields, regs, byte_size, info_line}} --- rbind_atoms = {[atom_name] = {binds, fields, regs, byte_size, info_line}}
--- rbind_structs = {[binds_name] = {byte_size, fields, atom_names}} --- rbind_structs = {[binds_name] = {byte_size, fields, atom_names}}
@@ -850,9 +880,9 @@ end
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)). --- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
--- ---
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization. --- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
--- @param corpus table -- the corpus from `ctx.shared.corpus` --- @param corpus table -- From `ctx.shared.corpus`
--- @param atom_table table[] -- the cross-ref'd atom table from build_atom_table --- @param atom_table table[] -- Cross-ref'd atom table from build_atom_table
--- @param registries table -- merged registries from collect_per_source_registries --- @param registries table -- Merged registries from collect_per_source_registries
--- @return table, table -- (rbind_atoms, rbind_structs) --- @return table, table -- (rbind_atoms, rbind_structs)
local function parse_rbind_atoms(corpus, atom_table, registries) local function parse_rbind_atoms(corpus, atom_table, registries)
registries = registries or {} registries = registries or {}
@@ -886,7 +916,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
local body_tokens_by_atom = {} local body_tokens_by_atom = {}
for _, src in ipairs((corpus and corpus.source_order) or {}) do for _, src in ipairs((corpus and corpus.source_order) or {}) do
local scan = src.scan local scan = src.scan
if scan then if scan then
for _, atom in ipairs(scan.atoms or {}) do for _, atom in ipairs(scan.atoms or {}) do
body_tokens_by_atom[atom.name] = atom.body_tokens body_tokens_by_atom[atom.name] = atom.body_tokens
end end
@@ -905,8 +935,8 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
for atom_name, ai in pairs(ai_by_atom) do for atom_name, ai in pairs(ai_by_atom) do
if ai.binds then if ai.binds then
local struct = rbind_structs[ai.binds] local struct = rbind_structs[ai.binds]
local body_toks = body_tokens_by_atom[atom_name] local body_toks = body_tokens_by_atom[atom_name]
if struct and body_toks then if struct and body_toks then
local pairs = parse_body_load_pairs(body_toks, ai.binds, registries) local pairs = parse_body_load_pairs(body_toks, ai.binds, registries)
if #pairs > 0 then if #pairs > 0 then
@@ -914,7 +944,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
binds = ai.binds, binds = ai.binds,
fields = struct.fields, -- {name, offset} from scan.binds fields = struct.fields, -- {name, offset} from scan.binds
bytes = struct.bytes, bytes = struct.bytes,
regs = pairs, -- ordered list of {reg, field} regs = pairs, -- Ordered list of {reg, field}
info_line = ai.info_line, info_line = ai.info_line,
} }
table.insert(struct.atom_names, atom_name) table.insert(struct.atom_names, atom_name)
@@ -941,7 +971,8 @@ end
--- (the final unit, referenced by the main CU's DW_AT_stmt_list). --- (the final unit, referenced by the main CU's DW_AT_stmt_list).
--- ---
--- This builder extends the main compilation unit. --- This builder extends the main compilation unit.
--- A detached synthetic line unit has no DW_AT_stmt_list referencing it, so gdb ignored it (a previous experiment); byte 13 is the first special opcode, not the extended-opcode marker. --- A detached synthetic line unit has no DW_AT_stmt_list referencing it, so gdb ignored it (a previous experiment);
--- byte 13 is the first special opcode, not the extended-opcode marker.
--- The existing final unit already contains hello_gte_tape.c as file index 11 and ends with a valid end_sequence. --- The existing final unit already contains hello_gte_tape.c as file index 11 and ends with a valid end_sequence.
--- We preserve its bytes, append independent atom sequences, and increase only that unit's DWARF32 unit_length. --- We preserve its bytes, append independent atom sequences, and increase only that unit's DWARF32 unit_length.
--- @param existing string -- existing section bytes, byte-for-byte --- @param existing string -- existing section bytes, byte-for-byte
@@ -952,9 +983,7 @@ local function build_dwarf_line_section(existing, atom_table)
-- Build the sequences. -- Build the sequences.
local sequences = {} local sequences = {}
for _, atom in ipairs(atom_table) do for _, atom in ipairs(atom_table) do sequences[#sequences + 1] = build_atom_sequence(atom) end
sequences[#sequences + 1] = build_atom_sequence(atom)
end
local appended = table.concat(sequences) local appended = table.concat(sequences)
-- Walk DWARF32 line units and retain the final unit's bounds. -- Walk DWARF32 line units and retain the final unit's bounds.
@@ -962,8 +991,8 @@ local function build_dwarf_line_section(existing, atom_table)
local unit_pos, last_pos, last_length, last_end = 0, nil, nil, nil local unit_pos, last_pos, last_length, last_end = 0, nil, nil, nil
while unit_pos < #existing do while unit_pos < #existing do
if unit_pos + 4 > #existing then return existing end if unit_pos + 4 > #existing then return existing end
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos) local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
if unit_length == elf_dwarf.ELF32.dw_dwarf32_terminator then return existing end if unit_length == elf_dwarf.dw_dwarf32_terminator then return existing end
local unit_end_excl = unit_pos + 4 + unit_length local unit_end_excl = unit_pos + 4 + unit_length
if unit_end_excl > #existing then return existing end if unit_end_excl > #existing then return existing end
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
@@ -1007,13 +1036,13 @@ local function build_dwarf_aranges_section(existing, atom_table)
-- We bump the unit's length field accordingly. -- We bump the unit's length field accordingly.
-- --
-- Unit structure (DWARF4 §7.21): -- Unit structure (DWARF4 §7.21):
-- unit_length (4) -- unit_length (4)
-- version (2) -- version (2)
-- debug_info_offset (4) -- CU DIE offset in .debug_info -- debug_info_offset (4) -- CU DIE offset in .debug_info
-- address_size (1) -- address_size (1)
-- segment_size (1) -- segment_size (1)
-- entries... (4-byte addr + 4-byte length) -- entries... (4-byte addr + 4-byte length)
-- terminator (8 bytes: addr=0, length=0) -- terminator (8 bytes: addr=0, length=0)
-- Walk all units and emit each one (preserving existing structure). -- Walk all units and emit each one (preserving existing structure).
-- For the LAST unit, replace the terminator with my entries + new term. -- For the LAST unit, replace the terminator with my entries + new term.
@@ -1024,7 +1053,7 @@ local function build_dwarf_aranges_section(existing, atom_table)
while i < #existing do while i < #existing do
-- Read this unit's length. -- Read this unit's length.
local ul = elf_dwarf.read_u32_le(existing, i) local ul = elf_dwarf.read_u32_le(existing, i)
if ul == elf_dwarf.ELF32.dw_dwarf32_terminator then if ul == elf_dwarf.dw_dwarf32_terminator then
-- DWARF64 marker - not supported. -- DWARF64 marker - not supported.
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n") io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
return existing return existing
@@ -1488,13 +1517,12 @@ end
--- DW_AT_location = piece-chain (DW_FORM_exprloc) --- DW_AT_location = piece-chain (DW_FORM_exprloc)
--- DW_AT_type = ref4 → structure_type DIE --- DW_AT_type = ref4 → structure_type DIE
--- ---
--- This function does NOT emit the final 0 byte (root terminator). build_debug_info_section splices bytes ahead of the root terminator --- This function does NOT emit the final 0 byte (root terminator).
--- and preserves existing DIE bytes exactly. --- build_debug_info_section splices bytes ahead of the root terminator and preserves existing DIE bytes exactly.
--- ---
--- ref4 basis: DW_FORM_ref4 is CU-relative (offset from the first byte of the CU header). --- ref4 basis: DW_FORM_ref4 is CU-relative (offset from the first byte of the CU header).
--- Our inserted DIEs live in the main CU, so every ref4 = (target section offset) - main_cu_offset. --- Our inserted DIEs live in the main CU, so every ref4 = (target section offset) - main_cu_offset.
--- Per-die section offsets are tracked via the running `next_offset` cursor (= section offset of the NEXT byte to emit). --- Per-die section offsets are tracked via the running `next_offset` cursor (= section offset of the NEXT byte to emit).
---
--- @param main_cu_offset integer -- 0-based section offset of the main CU's unit_length field --- @param main_cu_offset integer -- 0-based section offset of the main CU's unit_length field
--- @param main_cu_end_excl integer -- 0-based section offset of the first byte AFTER the main CU --- @param main_cu_end_excl integer -- 0-based section offset of the first byte AFTER the main CU
--- @param atom_table table[] -- atoms (with atom.rbind set if rbind; atom.invocations set if mac_X(...) calls) --- @param atom_table table[] -- atoms (with atom.rbind set if rbind; atom.invocations set if mac_X(...) calls)
@@ -1614,6 +1642,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
-- --
-- The table is small + explicit — the prototype principle treats the typed-view struct layout as data, not derived state. -- The table is small + explicit — the prototype principle treats the typed-view struct layout as data, not derived state.
local STRUCT_MEMBER_TABLE = { local STRUCT_MEMBER_TABLE = {
-- TODO(Ed): This hardcoding is brittle...
-- TODO(Ed): Better to just have a table for the fundamental types in duffle/dsl.h, we can derive the rest via typedef parsing...
-- 2-element signed short vector (rare; placeholder for future use). -- 2-element signed short vector (rare; placeholder for future use).
V2_S2 = { byte_size = 4, members = { V2_S2 = { byte_size = 4, members = {
{ name = "x", offset = 0, byte_size = 2 }, { name = "x", offset = 0, byte_size = 2 },
@@ -1750,7 +1780,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target) emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up. -- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type. The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below. type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type.
-- The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup. -- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding). -- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
local ptr_void_offset = void_chain_offset + 8 local ptr_void_offset = void_chain_offset + 8
@@ -1866,12 +1897,13 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
end end
local atom_view = (registries.atom_views or {})[atom.name] local atom_view = (registries.atom_views or {})[atom.name]
-- Build the atom-name lookup table once (cheap; O(atom_table)) so step (b) and step (d) can resolve rbind_atom names. -- Build the atom-name lookup table once (cheap; O(atom_table)) so step (b) and step (d) can resolve rbind_atom names.
-- TODO(Ed): Bad assignment?
local atom_by_name = atom_by_name or (function() local m = {}; for _, a in ipairs(atom_table) do if a.name then m[a.name] = a end end; return m end)() local atom_by_name = atom_by_name or (function() local m = {}; for _, a in ipairs(atom_table) do if a.name then m[a.name] = a end end; return m end)()
-- step (b) inputs: this atom's `atom_ctx(<rbind_atom>)` (resolved from the registries' atom_ctxs) -- step (b) inputs: this atom's `atom_ctx(<rbind_atom>)` (resolved from the registries' atom_ctxs)
local this_ctx = registries.atom_ctxs and registries.atom_ctxs[atom.name] local this_ctx = registries.atom_ctxs and registries.atom_ctxs[atom.name]
if this_ctx and this_ctx.rbind_atom then if this_ctx and this_ctx.rbind_atom then
local rbind = atom_by_name_global[this_ctx.rbind_atom] local rbind = atom_by_name_global[this_ctx.rbind_atom]
if rbind and rbind.rbind and rbind.rbind.fields then if rbind and rbind.rbind and rbind.rbind.fields then
atom_view_ctx_fields = {} atom_view_ctx_fields = {}
for _, f in ipairs(rbind.rbind.fields) do atom_view_ctx_fields[f.name] = f end for _, f in ipairs(rbind.rbind.fields) do atom_view_ctx_fields[f.name] = f end
if rbind.rbind.regs then if rbind.rbind.regs then
@@ -1891,11 +1923,11 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
end end
if my_phase_label then if my_phase_label then
local group = (registries.atom_phases or {})[my_phase_label] local group = (registries.atom_phases or {})[my_phase_label]
if group and group.atoms then if group and group.atoms then
for _, group_atom_name in ipairs(group.atoms) do for _, group_atom_name in ipairs(group.atoms) do
if group_atom_name ~= atom.name then if group_atom_name ~= atom.name then
local cand = atom_by_name_global[group_atom_name] local cand = atom_by_name_global[group_atom_name]
if cand and cand.rbind and cand.rbind.fields then if cand and cand.rbind and cand.rbind.fields then
atom_view_phase_fields = {} atom_view_phase_fields = {}
for _, f in ipairs(cand.rbind.fields) do atom_view_phase_fields[f.name] = f end for _, f in ipairs(cand.rbind.fields) do atom_view_phase_fields[f.name] = f end
if cand.rbind.regs then if cand.rbind.regs then
@@ -1916,7 +1948,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
-- (a) per-atom callsite atom_type(R_X, <T>): most specific; user explicit override for THIS atom only. -- (a) per-atom callsite atom_type(R_X, <T>): most specific; user explicit override for THIS atom only.
function(r_name, alias_code) function(r_name, alias_code)
local override = atom_view and atom_view.reg_type_overrides and atom_view.reg_type_overrides[r_name] local override = atom_view and atom_view.reg_type_overrides and atom_view.reg_type_overrides[r_name]
if override and override.pointer_depth and override.pointer_depth > 0 then if override and override.pointer_depth and override.pointer_depth > 0 then
return type_chain_offsets[override.type_name .. "|" .. override.pointer_depth] return type_chain_offsets[override.type_name .. "|" .. override.pointer_depth]
end end
end, end,
@@ -1924,7 +1956,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
function(r_name, alias_code) function(r_name, alias_code)
local ctx_field_name = reg_to_field_ctx and reg_to_field_ctx[alias_code] local ctx_field_name = reg_to_field_ctx and reg_to_field_ctx[alias_code]
local ctx_f = ctx_field_name and atom_view_ctx_fields and atom_view_ctx_fields[ctx_field_name] local ctx_f = ctx_field_name and atom_view_ctx_fields and atom_view_ctx_fields[ctx_field_name]
if ctx_f and ctx_f.pointer_depth and ctx_f.pointer_depth > 0 then if ctx_f and ctx_f.pointer_depth and ctx_f.pointer_depth > 0 then
return type_chain_offsets[ctx_f.type_name .. "|" .. ctx_f.pointer_depth] return type_chain_offsets[ctx_f.type_name .. "|" .. ctx_f.pointer_depth]
end end
end, end,
@@ -1932,7 +1964,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
function(r_name, alias_code) function(r_name, alias_code)
local field_name = reg_to_field[alias_code] local field_name = reg_to_field[alias_code]
local f = field_name and field_type_by_name[field_name] local f = field_name and field_type_by_name[field_name]
if f and f.pointer_depth and f.pointer_depth > 0 then if f and f.pointer_depth and f.pointer_depth > 0 then
return type_chain_offsets[f.type_name .. "|" .. f.pointer_depth] return type_chain_offsets[f.type_name .. "|" .. f.pointer_depth]
end end
end, end,
@@ -1940,12 +1972,13 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
function(r_name, alias_code) function(r_name, alias_code)
local phase_field_name = reg_to_field_phase and reg_to_field_phase[alias_code] local phase_field_name = reg_to_field_phase and reg_to_field_phase[alias_code]
local phase_f = phase_field_name and atom_view_phase_fields and atom_view_phase_fields[phase_field_name] local phase_f = phase_field_name and atom_view_phase_fields and atom_view_phase_fields[phase_field_name]
if phase_f and phase_f.pointer_depth and phase_f.pointer_depth > 0 then if phase_f and phase_f.pointer_depth and phase_f.pointer_depth > 0 then
return type_chain_offsets[phase_f.type_name .. "|" .. phase_f.pointer_depth] return type_chain_offsets[phase_f.type_name .. "|" .. phase_f.pointer_depth]
end end
end, end,
-- (e) enum-site atom_type(<T>) default on the registry entry. -- (e) enum-site atom_type(<T>) default on the registry entry.
function(r_name, alias_code) function(r_name, alias_code)
-- TODO(Ed): Bad definition?
if alias and alias.default_type and alias.default_depth and alias.default_depth > 0 then if alias and alias.default_type and alias.default_depth and alias.default_depth > 0 then
return type_chain_offsets[alias.default_type .. "|" .. alias.default_depth] return type_chain_offsets[alias.default_type .. "|" .. alias.default_depth]
end end
@@ -1954,8 +1987,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
-- Iterate `by_alias` in sorted order; Lua's pairs() is non-deterministic, so sorting ensures byte-identical DWARF output across builds. -- Iterate `by_alias` in sorted order; Lua's pairs() is non-deterministic, so sorting ensures byte-identical DWARF output across builds.
for _, r_name in ipairs(by_alias_order) do for _, r_name in ipairs(by_alias_order) do
local alias = by_alias[r_name] local alias = by_alias[r_name]
local rr_name = "RR_" .. strip_r_prefix(r_name) local rr_name = "RR_" .. strip_r_prefix(r_name)
local alias_code = alias.code local alias_code = alias.code
emit(uleb128(ABBREV_VARIABLE)) emit(uleb128(ABBREV_VARIABLE))
emit(rr_name .. "\0") -- DW_FORM_string (DW_AT_name) emit(rr_name .. "\0") -- DW_FORM_string (DW_AT_name)
@@ -1977,7 +2010,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
-- Two PC ranges cover every field: [atom.addr, last_load+8) describes each field as tape memory (DW_OP_bregN + offset) piece, -- Two PC ranges cover every field: [atom.addr, last_load+8) describes each field as tape memory (DW_OP_bregN + offset) piece,
-- and [last_load+8, atom.end) describes each field as a GPR (DW_OP_regN) piece. -- and [last_load+8, atom.end) describes each field as a GPR (DW_OP_regN) piece.
if atom.rbind then if atom.rbind then
local binds_name = atom.rbind.binds local binds_name = atom.rbind.binds
local loclists_offset = loclists_offsets[atom.name] or 0 local loclists_offset = loclists_offsets[atom.name] or 0
emit(uleb128(ABBREV_BIND_VAR_LOCLIST)) emit(uleb128(ABBREV_BIND_VAR_LOCLIST))
emit("bind_args\0") -- DW_FORM_string (DW_AT_name) emit("bind_args\0") -- DW_FORM_string (DW_AT_name)
@@ -2030,7 +2063,7 @@ end
--- ---
--- Fails safely by returning existing sections unchanged if the table walker can't find the table terminator (malformed input). --- Fails safely by returning existing sections unchanged if the table walker can't find the table terminator (malformed input).
--- ---
--- @param existing string -- existing .debug_abbrev bytes, byte-for-byte --- @param existing string -- existing .debug_abbrev bytes, byte-for-byte
--- @param main_abbrev_offset integer -- 0-based offset into `existing` of the main CU's abbrev table --- @param main_abbrev_offset integer -- 0-based offset into `existing` of the main CU's abbrev table
--- @return string, integer -- (new_abbrev_bytes, offset_where_duplicate_table_starts = #existing) --- @return string, integer -- (new_abbrev_bytes, offset_where_duplicate_table_starts = #existing)
local function build_debug_abbrev_section(existing, main_abbrev_offset) local function build_debug_abbrev_section(existing, main_abbrev_offset)
@@ -2048,7 +2081,7 @@ local function build_debug_abbrev_section(existing, main_abbrev_offset)
end end
--- Build the new .debug_str: existing strings + new strings appended. --- Build the new .debug_str: existing strings + new strings appended.
--- @param existing string -- existing .debug_str bytes, byte-for-byte --- @param existing string -- existing .debug_str bytes, byte-for-byte
--- @param atom_table table[] --- @param atom_table table[]
--- @param registries table -- merged registries from collect_per_source_registries --- @param registries table -- merged registries from collect_per_source_registries
--- @return string -- existing bytes plus the deterministic appended strings --- @return string -- existing bytes plus the deterministic appended strings
@@ -2058,7 +2091,6 @@ local function build_debug_str_section(existing, atom_table, registries)
end end
--- Build the new .debug_info: SPLICE inserted DIEs into the MAIN CU as children. --- Build the new .debug_info: SPLICE inserted DIEs into the MAIN CU as children.
---
--- This implementation: --- This implementation:
--- 1. Builds the inserted-children bytes (base_type, struct_types, subprograms with their RR_* + bind_args children) via build_inserted_children. --- 1. Builds the inserted-children bytes (base_type, struct_types, subprograms with their RR_* + bind_args children) via build_inserted_children.
--- 2. Patches the main CU's `unit_length` field to account for the inserted bytes. --- 2. Patches the main CU's `unit_length` field to account for the inserted bytes.
@@ -2068,14 +2100,14 @@ end
--- ---
--- The crt CU (everything before main_cu_start) is preserved. --- The crt CU (everything before main_cu_start) is preserved.
--- @param existing string -- existing .debug_info section bytes --- @param existing string -- existing .debug_info section bytes
--- @param main_cu_start integer -- 0-based offset of the main CU's unit_length field --- @param main_cu_start integer -- 0-based offset of the main CU's unit_length field
--- @param main_cu_end_excl integer -- 0-based offset of the first byte AFTER the main CU --- @param main_cu_end_excl integer -- 0-based offset of the first byte AFTER the main CU
--- @param new_abbrev_offset integer -- 0-based offset into the new .debug_abbrev of the duplicate main table --- @param new_abbrev_offset integer -- 0-based offset into the new .debug_abbrev of the duplicate main table
--- @param atom_table table[] --- @param atom_table table[]
--- @param rbind_structs table -- {[binds_name] = {bytes, fields, atom_names}} --- @param rbind_structs table -- {[binds_name] = {bytes, fields, atom_names}}
--- @param loclists_offsets table -- {[atom_name] = section-relative offset} --- @param loclists_offsets table -- {[atom_name] = section-relative offset}
--- @param registries table -- merged registries from collect_per_source_registries --- @param registries table -- merged registries from collect_per_source_registries
--- @return string -- the rebuilt .debug_info bytes --- @return string -- the rebuilt .debug_info bytes
local function build_debug_info_section(existing, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries) local function build_debug_info_section(existing, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries)
-- 1) Build the inserted children bytes (just before the main CU's root terminator). -- 1) Build the inserted children bytes (just before the main CU's root terminator).
@@ -2133,13 +2165,13 @@ local SECTION_WRITERS = {
-- Write a list of `{name, data}` section records to disk via SECTION_WRITERS. -- Write a list of `{name, data}` section records to disk via SECTION_WRITERS.
-- @param results table[] -- list of `{name=, data=}` records to write -- @param results table[] -- list of `{name=, data=}` records to write
-- @param ctx PassCtx -- @param ctx PassCtx
-- @param basename string -- output file basename (e.g. "hello_gte") -- @param basename string -- output file basename (e.g. "hello_gte")
-- @return table -- list of {name_bin = path} entries to append to M.run's outputs -- @return table -- list of {name_bin = path} entries to append to M.run's outputs
local function write_sections(results, ctx, basename) local function write_sections(results, ctx, basename)
local outputs = {} local outputs = {}
for _, r in ipairs(results) do for _, r in ipairs(results) do
local path = SECTION_WRITERS[r.name](ctx.out_root, basename) local path = SECTION_WRITERS[r.name](ctx.out_root, basename)
local f = io.open(path, "wb") local f = io.open(path, "wb")
if not f then if not f then
io.stderr:write(string.format("[dwarf_injection] failed to open %s for write\n", path)) io.stderr:write(string.format("[dwarf_injection] failed to open %s for write\n", path))
else else
@@ -2166,7 +2198,7 @@ function M.run(ctx)
end end
-- Guard: --elf is required. -- Guard: --elf is required.
local elf_path = ctx.flags and ctx.flags.elf_path local elf_path = ctx.flags and ctx.flags.elf_path
if not elf_path or elf_path == "" then if not elf_path or elf_path == "" then
io.stderr:write("[dwarf_injection] --elf flag missing\n") io.stderr:write("[dwarf_injection] --elf flag missing\n")
return { outputs = {}, errors = {}, warnings = {} } return { outputs = {}, errors = {}, warnings = {} }
@@ -2177,7 +2209,8 @@ function M.run(ctx)
-- Read the existing DWARF sections directly (no subprocess; io.open + manual ELF32 section-header walk). -- Read the existing DWARF sections directly (no subprocess; io.open + manual ELF32 section-header walk).
-- We need all 8 sections: .debug_line / .debug_aranges / .debug_rnglists get extended (additional rows appended to the existing unit), -- We need all 8 sections: .debug_line / .debug_aranges / .debug_rnglists get extended (additional rows appended to the existing unit),
-- and .debug_info / .debug_abbrev / .debug_str / .debug_loc / .debug_loclists get spliced (the main CU's unit_length is patched; no new compile unit is appended; .debug_loc/.debug_loclists may not exist in the source ELF so we add-section them on splice). -- and .debug_info / .debug_abbrev / .debug_str / .debug_loc / .debug_loclists get spliced
-- (the main CU's unit_length is patched; no new compile unit is appended; .debug_loc/.debug_loclists may not exist in the source ELF so we add-section them on splice).
-- The per-section dispatch is inlined in the writers loop below. -- The per-section dispatch is inlined in the writers loop below.
local existing_sections = elf_dwarf.read_elf_sections(elf_path, { local existing_sections = elf_dwarf.read_elf_sections(elf_path, {
".debug_line", ".debug_aranges", ".debug_rnglists", ".debug_line", ".debug_aranges", ".debug_rnglists",
@@ -2186,14 +2219,17 @@ function M.run(ctx)
-- reading them just returns "" which is the "missing" case the builder handles. -- reading them just returns "" which is the "missing" case the builder handles.
".debug_loc", ".debug_loclists", ".debug_loc", ".debug_loclists",
}) })
-- Resolve the per-file line-table indices from the same .debug_line bytes;
-- this MUST run before any atom sequence is emitted (build_atom_sequence below
-- calls resolve_provenance_file_index when populating call-site / body rows).
init_file_index_lookup(elf_path)
-- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation). -- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation).
-- `corpus` is the sole canonical source projection. -- `corpus` is the sole canonical source projection.
local corpus = (ctx.shared and ctx.shared.corpus) or {} local corpus = (ctx.shared and ctx.shared.corpus) or {}
local registries = collect_per_source_registries(corpus) local registries = collect_per_source_registries(corpus)
-- Read nm symbols (the ONLY disk-side input to the atom table) and join -- Read nm symbols (the ONLY disk-side input to the atom table) and join them against `corpus.atoms_by_name` + `atom.paths` for word rows + invocation ancestry.
-- them against `corpus.atoms_by_name` + `atom.paths` for word rows + invocation ancestry.
-- Disk source-map/provenance text is not consulted (those are diagnostic artifacts; semantic inputs are in memory). -- Disk source-map/provenance text is not consulted (those are diagnostic artifacts; semantic inputs are in memory).
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path) local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
local atom_table = build_atom_table(corpus, addrs) local atom_table = build_atom_table(corpus, addrs)
-- Detect rbind atoms + index Binds_* struct fields from the corpus. -- Detect rbind atoms + index Binds_* struct fields from the corpus.
@@ -2217,7 +2253,7 @@ function M.run(ctx)
duffle.ensure_dir(ctx.out_root) duffle.ensure_dir(ctx.out_root)
-- Step 0: layout validation. Bail out safely if the .debug_info layout doesn't match what we expect (crt CU + DWARF5 main CU + final 0 byte). -- Step 0: layout validation. Bail out safely if the .debug_info layout doesn't match what we expect (crt CU + DWARF5 main CU + final 0 byte).
-- A layout mismatch means the gcc emission changed; the safest response is to leave existing sections unchanged and emit no synthetic data, so the build's debug-info step never silently produces broken DWARF. -- A layout mismatch means the gcc emission changed; the safest response is to leave existing sections unchanged and emit no synthetic data, so the build's debug-info step never silently produces broken DWARF.
local existing_info = existing_sections[".debug_info"] or "" local existing_info = existing_sections[".debug_info"] or ""
local existing_abbrev = existing_sections[".debug_abbrev"] or "" local existing_abbrev = existing_sections[".debug_abbrev"] or ""
local main_cu_start, main_cu_end_excl, main_abbrev_offset = find_main_cu_layout(existing_info) local main_cu_start, main_cu_end_excl, main_abbrev_offset = find_main_cu_layout(existing_info)
@@ -2252,7 +2288,7 @@ function M.run(ctx)
local new_info = build_debug_info_section(existing_info, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries) local new_info = build_debug_info_section(existing_info, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries)
-- Step 2b: rebuild .debug_str now that we know which RR_<R_Name> entries get emitted. -- Step 2b: rebuild .debug_str now that we know which RR_<R_Name> entries get emitted.
-- This aligns with build_debug_info_section's by_alias loop. -- This aligns with build_debug_info_section's by_alias loop.
local new_str = build_debug_str_section(existing_sections[".debug_str"] or "", atom_table, registries) local new_str = build_debug_str_section(existing_sections[".debug_str"] or "", atom_table, registries)
-- Step 3-5: independent sections. -- Step 3-5: independent sections.
+10 -12
View File
@@ -41,7 +41,6 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Convert the recursive walk's body-relative line numbers into physical source lines once. -- Convert the recursive walk's body-relative line numbers into physical source lines once.
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`. -- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure. -- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
--
-- `call_line` discipline: -- `call_line` discipline:
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker. -- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once. -- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
@@ -59,10 +58,9 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`. -- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`. -- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line. -- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
local root_body_line = root_line_of(atom_record.body_off - 1) local root_body_line = root_line_of(atom_record.body_off - 1) or atom_record.line or 0
or atom_record.line or 0
local component_index = corpus.component_body_index or {} local component_index = corpus.component_body_index or {}
local word_items = {} local word_items = {}
for _, item in ipairs(projection.items) do for _, item in ipairs(projection.items) do
if item.kind == "word" then word_items[#word_items + 1] = item end if item.kind == "word" then word_items[#word_items + 1] = item end
@@ -102,7 +100,7 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
end end
-- Normalize `inv.call_line` to a physical source line. -- Normalize `inv.call_line` to a physical source line.
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`. -- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged. -- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
for _, inv in ipairs(projection.invocations) do for _, inv in ipairs(projection.invocations) do
if inv.parent_id == 0 then if inv.parent_id == 0 then
@@ -114,14 +112,14 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here. -- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged. -- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
for _, inv in ipairs(projection.invocations) do for _, inv in ipairs(projection.invocations) do
local sw = inv.start_word local sw = inv.start_word
local ew = inv.end_word local ew = inv.end_word
local bls = {} local bls = {}
for i = sw, ew do for i = sw, ew do
local it = projection.items and projection.items[i] local it = projection.items and projection.items[i]
if it and it.kind == "word" then if it and it.kind == "word" then
local fake_event = { invocation_ids = { inv.id } } local fake_event = { invocation_ids = { inv.id } }
bls[#bls + 1] = body_line_for(fake_event, it) or 0 bls[#bls + 1] = body_line_for(fake_event, it) or 0
end end
end end
inv.body_lines = bls inv.body_lines = bls
@@ -190,11 +188,11 @@ function M.run(ctx)
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
-- Project once, collect errors + warnings for one atom. -- Project once, collect errors + warnings for one atom.
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc. -- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
local function process_atom(atom, src) local function process_atom(atom, src)
if not (atom and atom.body) then return end if not (atom and atom.body) then return end
local kind = atom.kind local kind = atom.kind
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
return return
end end
local proj = project_atom(atom, src, corpus) local proj = project_atom(atom, src, corpus)
@@ -217,7 +215,7 @@ function M.run(ctx)
end end
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms. -- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission. -- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly. -- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
for _, src in ipairs(corpus.source_order) do for _, src in ipairs(corpus.source_order) do
local scan = src.scan or {} local scan = src.scan or {}
+104 -59
View File
@@ -3,12 +3,19 @@
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`) --- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset --- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits --- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
--- `<dir_basename>.offsets.h` with one `#define _atom_offset_F_T = N` per branch. --- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
---
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
--- The directory itself is the namespace; the filename does not repeat the module name.
---
--- (Task 12.16 note: atom-namespaced enum names — e.g., `atom_offset__normalize_v3s4__srav_path__aligned_done` —
--- were considered to prevent cross-atom label collisions, but the C-side `atom_offset(F, T)` macro in
--- `code/duffle/dsl.atom.h` doesn't know the current atom_name at expansion time, so any namespacing
--- on the metaprogram side breaks the C build. Reverted. The C-side would need a per-atom
--- `CURRENT_ATOM` #define (set by `MipsAtom_`/`MipsAtom_Proc_` macros) plus an updated `atom_offset`
--- macro that uses it. That's a coordinated refactor — deferred to a future track.)
--- ---
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1). --- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -16,12 +23,11 @@
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale. -- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
-- Uses `debug.getinfo` to find this file's own directory, so it works -- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- both standalone and when require'd from the orchestrator.
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd). -- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -39,40 +45,42 @@ local OFFSET_MACRO_COL = 44
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- Absolute path to the source file
--- @field text string -- the full source text --- @field text string -- Full source text
--- @field dir string -- the directory containing the source --- @field dir string -- Directory containing the source
--- @field basename string -- filename without extension --- @field basename string -- Filename without extension
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source) --- @field scan table -- Pre-scanned SourceScan payload (from duffle.scan_source)
--- @class PassCtx --- @class PassCtx
--- @field shared table -- cross-pass shared state --- @field shared table -- Cross-pass shared state
--- @field shared.corpus table -- canonical corpus projection --- @field shared.corpus table -- Corpus projection
--- @field shared.word_counts table --- @field shared.word_counts table
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- Output root (e.g. "build/gen")
--- @class PassResult --- @class PassResult
--- @field outputs table[] -- {kind=, path=} entries describing emit files --- @field outputs table[] -- {kind=, path=} entries describing emit files
--- @field errors table[] -- {line=, msg=} entries; build-stops --- @field errors table[] -- {line=, msg=} entries; build-stops
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds --- @field warnings table[] -- {line=, msg=} entries; build-succeeds
--- @class BranchOffset --- @class BranchOffset
--- @field tag string -- the marker tag (e.g. "F" in `atom_offset(F, T)`) --- @field tag string -- Marker tag (e.g. "F" in `atom_offset(F, T)`)
--- @field target string -- the target label name (e.g. "T" in `atom_offset(F, T)`) --- @field target string -- Target label name (e.g. "T" in `atom_offset(F, T)`)
--- @field branch_word integer -- branch word position within the atom body --- @field branch_word integer -- Branch word position within the atom body
--- @field offset integer -- computed `target_word - branch_word - 1` --- @field offset integer -- Computed per consuming instruction (see `compute_offsets`)
--- @field consuming_encoder string|nil -- Instruction consuming the offset (e.g. "branch_le_zero", "jump", "call_addr")
--- @field consuming_arg_pos integer|nil -- 1-based arg position within the consuming instruction's arg list
--- @class AtomData --- @class AtomData
--- @field name string -- atom name --- @field name string -- Atom name
--- @field total_words integer -- total word count of the atom body --- @field total_words integer -- Total word count of the atom body
--- @field offsets BranchOffset[] -- per-branch offset list --- @field offsets BranchOffset[] -- Per-branch offset list
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Canonical marker projection -- Canonical marker projection
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- MARKER_PROJECTORS is the marker-kind data table. -- MARKER_PROJECTORS is the marker-kind data table.
-- The emission-model pass already records marker word positions; -- The emission-model pass already records marker word positions + consuming-instruction context;
-- this pass only projects those records into the label/branch lookup shape needed by offset computation. -- this pass only projects those records into the label/branch lookup shape needed by offset computation.
local MARKER_PROJECTORS = { local MARKER_PROJECTORS = {
label = function(state, marker) label = function(state, marker)
@@ -80,9 +88,11 @@ local MARKER_PROJECTORS = {
end, end,
offset = function(state, marker) offset = function(state, marker)
state.branches[#state.branches + 1] = { state.branches[#state.branches + 1] = {
tag = marker.name, tag = marker.name,
target = marker.target, target = marker.target,
branch_word = marker.word_index, branch_word = marker.word_index,
consuming_encoder = marker.consuming_encoder,
consuming_arg_pos = marker.consuming_arg_pos,
} }
end, end,
} }
@@ -95,7 +105,7 @@ local function project_markers(markers)
local state = { labels = {}, branches = {} } local state = { labels = {}, branches = {} }
for _, marker in ipairs(markers or {}) do for _, marker in ipairs(markers or {}) do
local project = MARKER_PROJECTORS[marker.kind] local project = MARKER_PROJECTORS[marker.kind]
if project then project(state, marker) end if project then project(state, marker) end
end end
return state.labels, state.branches return state.labels, state.branches
end end
@@ -104,8 +114,19 @@ end
-- Offset computation + header generation -- Offset computation + header generation
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- Compute branch offsets as `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding). --- Compute branch offsets per consuming instruction.
--- @param labels table<string, integer> --- Disposition table:
--- `branch_*` -> relative offset: `target_word - branch_word - 1` (MIPS branch-immediate encoding).
--- `jump` / `call_addr` -> same value as `branch_*` (a relative word offset).
--- The duffle headers' `enc_i` macro truncates the value to the immediate-field width (16 bits for branches, 26 bits for jumps).
--- For tape-atom bodies within a single module, this works for `j`/`jal` because the linker's symbol resolution produces the correct 26-bit absolute target via standard `j` relocations.
--- For cross-module `j`/`jal` (atom body in one module, target in another), the linker emits a `R_MIPS_26` relocation against the lower 26 bits; the upper 4 bits come from the PC of the delay slot following the `j`.
--- The metaprogram doesn't know either at compile time, so the emitted value is the relative word offset that the duffle `enc_i` macro places in the immediate field; the toolchain handles the rest.
--- `jump_reg` / `call_reg` / `jump_link` -> ERROR. Register-form jumps have no offset field; `atom_offset` is invalid.
---
--- Top-level `atom_offset(F, T)` markers (where the marker is the entire token — `consuming_encoder` == nil) default to `branch_*` behavior (relative offset).
--- This preserves backward compatibility for any top-level marker that may exist outside a control-transfer instruction.
--- @param labels table<string, integer>
--- @param branches table[] --- @param branches table[]
--- @return BranchOffset[] --- @return BranchOffset[]
local function compute_offsets(labels, branches) local function compute_offsets(labels, branches)
@@ -115,11 +136,23 @@ local function compute_offsets(labels, branches)
if not target then if not target then
error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.branch_word .. ")") error("Branch target '" .. br.target .. "' has no atom_label (at word " .. br.branch_word .. ")")
end end
local consuming = br.consuming_encoder
local offset
if consuming == "jump_reg" or consuming == "call_reg" or consuming == "jump_link" then
-- Register-form jumps have no offset field. `atom_offset` cannot be used here.
error("atom_offset cannot be used with " .. consuming
.. " (register-form jumps have no offset field); at word " .. br.branch_word)
end
-- All other consuming instructions (including `branch_*`, `jump`, `call_addr`, and nil for top-level markers) use the same relative offset value.
-- The MIPS encoding differs per opcode but the duffle `enc_i` macro handles the truncation to the immediate-field width.
offset = target - br.branch_word - 1
results[#results + 1] = { results[#results + 1] = {
target = br.target, target = br.target,
tag = br.tag, tag = br.tag,
branch_word = br.branch_word, branch_word = br.branch_word,
offset = target - br.branch_word - 1, offset = offset,
consuming_encoder = br.consuming_encoder,
consuming_arg_pos = br.consuming_arg_pos,
} }
end end
return results return results
@@ -167,44 +200,48 @@ local function emit_atom_offsets(add, atom)
add("") add("")
end end
--- Generate the per-source .offsets.h header. --- Generate the per-directory .offsets.h header.
--- @param source_path string --- @param dir string -- the absolute source directory
--- @param atoms_data AtomData[] --- @param sources table[] -- sources contributing to this directory (for the header comment)
--- @param atoms_data AtomData[]
--- @return string --- @return string
local function generate_header(source_path, atoms_data) local function generate_header(dir, sources, atoms_data)
local basename = duffle.basename_no_ext(source_path) local dir_basename = duffle.basename_no_ext(dir)
local lines = {} local lines = {}
local function add(s) lines[#lines + 1] = s end local function add(s) lines[#lines + 1] = s end
add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT") add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT")
add("// Source: " .. source_path) add("// Directory: " .. dir:gsub("/", "\\") .. "\\")
for _, src in ipairs(sources) do
add("// source: " .. src.path:gsub("/", "\\"))
end
add("#pragma once") add("#pragma once")
add("") add("")
add("#pragma region " .. basename) add("#pragma region " .. dir_basename)
add("") add("")
add("") add("")
for _, atom in ipairs(atoms_data) do for _, atom in ipairs(atoms_data) do
emit_atom_offsets(add, atom) emit_atom_offsets(add, atom)
end end
add("#pragma endregion " .. basename) add("#pragma endregion " .. dir_basename)
add("") add("")
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
local M = {} local M = {}
--- (internal) Process one source: render offsets from canonical atom paths. --- (internal) Aggregate atoms from every source in one directory, render the per-directory `offsets.h`.
--- Returns the offsets_h path if a header was written, or nil. --- Returns the offsets_h path if a header was written, or nil.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @param src SourceFile --- @param dir string -- the absolute source directory
--- @param sources SourceFile[] -- sources in this directory
--- @return string|nil -- the offsets_h path --- @return string|nil -- the offsets_h path
local function process_source(ctx, src) local function process_directory(ctx, dir, sources)
local atoms_data = {} local atoms_data = {}
local scan = src.scan or {}
local function append_atom(atom) local function append_atom(atom)
local paths = atom and atom.paths local paths = atom and atom.paths
if not paths then return end if not paths then return end
local labels, branches = project_markers(paths.markers) local labels, branches = project_markers(paths.markers)
atoms_data[#atoms_data + 1] = { atoms_data[#atoms_data + 1] = {
@@ -214,19 +251,22 @@ local function process_source(ctx, src)
} }
end end
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end for _, src in ipairs(sources) do
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end local scan = src.scan or {}
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end
end
if #atoms_data == 0 then return nil end if #atoms_data == 0 then return nil end
local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h" local out_path = dir .. "/gen/offsets.h"
duffle.ensure_dir(duffle.dirname(out_path)) duffle.ensure_dir(duffle.dirname(out_path))
duffle.write_file(out_path, generate_header(src.path:gsub("/", "\\"), atoms_data)) duffle.write_file(out_path, generate_header(dir, sources, atoms_data))
return out_path return out_path
end end
--- Run the offsets pass. --- Run the offsets pass.
--- For each canonical source, emits a per-module `<dir_basename>.offsets.h` --- For each canonical source-directory, emits a per-directory `gen/offsets.h`
--- containing constants for every marker recorded in atom.paths. --- containing constants for every marker recorded in atom.paths across every source in that directory.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
@@ -235,12 +275,17 @@ function M.run(ctx)
local warnings = {} local warnings = {}
local corpus = ctx.shared and ctx.shared.corpus local corpus = ctx.shared and ctx.shared.corpus
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then if type(corpus) ~= "table" then
error("offsets.run requires ctx.shared.corpus.source_order (canonical corpus).", 0) error("offsets.run requires ctx.shared.corpus", 0)
end
if type(corpus.source_order) ~= "table" then
error("offsets.run requires ctx.shared.corpus.source_order.", 0)
end end
for _, src in ipairs(corpus.source_order) do -- Per-directory aggregation: every source in the same directory contributes to one `gen/offsets.h`.
local out_path = process_source(ctx, src) local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order)
for dir, sources in pairs(sources_by_dir) do
local out_path = process_directory(ctx, dir, sources)
if out_path then if out_path then
outputs[#outputs + 1] = { offsets_h = out_path } outputs[#outputs + 1] = { offsets_h = out_path }
end end
+463 -339
View File
@@ -1,34 +1,34 @@
--- passes/report.lua — Per-MODULE annotation report renderer + --- passes/report.lua — Per-MODULE annotation report renderer + project-wide summary writer.
--- project-wide summary writer.
--- ---
--- Two output files per build: --- Two output files per build:
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory. --- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
--- - `build/gen/annotation_validation.txt` — the project summary. --- - `build/gen/annotation_validation.txt` — the project summary.
--- ---
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory. --- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results. --- This pass iterates the dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD. -- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
-- Bootstrap: see `ps1_meta.lua` for the rationale. -- Bootstrap: See `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- Bootstrap: Load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
-- Uses `debug.getinfo` to find this file's own directory, so it works -- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- both standalone and when require'd from the orchestrator. -- Bootstrap: Load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection. -- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings) -- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output. -- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
local annotation = dofile(_bootstrap_dir .. "annotation.lua") local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
local atoms_source_map = dofile(_bootstrap_dir .. "atoms_source_map.lua")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -36,7 +36,7 @@ local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- Section separators used in the rendered text reports. -- Section separators used in the rendered text reports.
-- The thin rules are hand-tuned to align with the per-section content width; do not change without also checking the section renderers below. -- The thin rules are hand-tuned to align with the per-section content width; do not change without also checking the section renderers below.
local RULE_THICK = "========================================================" local RULE_THICK = "========================================================"
local SECTION_HEADER_ATOMS = "── Atoms ────────────────────────────────────────────────" local SECTION_HEADER_ATOMS = "── Atoms ────────────────────────────────────────────────"
local SECTION_HEADER_ANNOTS = "── Annotations ──────────────────────────────────────────" local SECTION_HEADER_ANNOTS = "── Annotations ──────────────────────────────────────────"
local SECTION_HEADER_BINDS = "── Binds_* structs ──────────────────────────────────────" local SECTION_HEADER_BINDS = "── Binds_* structs ──────────────────────────────────────"
@@ -44,8 +44,7 @@ local SECTION_HEADER_MACROS = "── Macro word-count declarations ───
local SECTION_HEADER_ERRORS = "── Errors ──────────────────────────────────────────────" local SECTION_HEADER_ERRORS = "── Errors ──────────────────────────────────────────────"
local SECTION_HEADER_WARNINGS = "── Warnings ────────────────────────────────────────────" local SECTION_HEADER_WARNINGS = "── Warnings ────────────────────────────────────────────"
-- Lua pattern that captures the basename (last path segment) of a -- Lua pattern that captures the basename (last path segment) of a forward- or back-slash separated path.
-- forward- or back-slash separated path.
local BASENAME_PATTERN = "([^/\\]+)$" local BASENAME_PATTERN = "([^/\\]+)$"
-- Debug flag name — set to truthy in `_G` to enable verbose logging. -- Debug flag name — set to truthy in `_G` to enable verbose logging.
@@ -59,83 +58,83 @@ local PASS_NAME = "report"
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- Absolute path to the source file
--- @field text string -- the full source text --- @field text string -- Full source text
--- @field dir string -- the directory containing the source --- @field dir string -- Directory containing the source
--- @field basename string -- filename without extension --- @field basename string -- Filename without extension
--- @class PassCtx --- @class PassCtx
--- @field sources SourceFile[] -- all source files in the build --- @field sources SourceFile[] -- All source files in the build
--- @field metadata_path string -- path to word_count.metadata.h --- @field metadata_path string -- Path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- Cross-pass shared state
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- Output root (e.g. "build/gen")
--- @field project_root string -- project root (e.g. "code/") --- @field project_root string -- Project root (e.g. "code/")
--- @field upstream table<string, table> -- per-pass upstream outputs --- @field upstream table<string, table> -- Per-pass upstream outputs
--- @field flags table -- CLI flags + per-pass stash --- @field flags table -- CLI flags + per-pass stash
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- If true, log diagnostic info
--- @class PassResult --- @class PassResult
--- @field outputs table[] -- {kind=, path=} entries describing emit files --- @field outputs table[] -- {kind=, path=} entries describing emit files
--- @field errors table[] -- {line=, msg=} entries; build-stops --- @field errors table[] -- {line=, msg=} entries; build-stops
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds --- @field warnings table[] -- {line=, msg=} entries; build-succeeds
-- Shapes produced by `passes/annotation.lua`'s `M.validate()`. -- Shapes produced by `passes/annotation.lua`'s `M.validate()`.
--- @class AtomEntry --- @class AtomEntry
--- @field name string -- atom name (e.g. "cube_g4_face") --- @field name string -- Atom name (e.g. "cube_g4_face")
--- @field line integer -- source line of the atom declaration --- @field line integer -- Source line of the atom declaration
--- @class AnnotEntry --- @class AnnotEntry
--- @field line integer -- source line --- @field line integer -- Source line
--- @field macro string -- the macro name (e.g. "atom_reads") --- @field macro string -- Macro name (e.g. "atom_reads")
--- @field name string -- the atom name (if a `name(...)` was given) --- @field name string -- Atom name (if a `name(...)` was given)
--- @field kind string -- "atom_info" | "atom_bind" | ... --- @field kind string -- "atom_info" | "atom_bind" | ...
--- @field binds string|nil -- Binds_X name if any --- @field binds string|nil -- Binds_X name if any
--- @field reads string[] -- R_* names (read targets) --- @field reads string[] -- R_* names (read targets)
--- @field writes string[] -- R_* names (write targets) --- @field writes string[] -- R_* names (write targets)
--- @field error string|nil -- error message if annotation was malformed --- @field error string|nil -- Error message if annotation was malformed
--- @class BindsField --- @class BindsField
--- @field name string -- field name --- @field name string -- Field name
--- @field offset integer -- byte offset within the Binds_X struct --- @field offset integer -- Byte offset within the Binds_X struct
--- @class BindsStruct --- @class BindsStruct
--- @field name string -- struct name (e.g. "Binds_Floor") --- @field name string -- Struct name (e.g. "Binds_Floor")
--- @field line integer -- source line of the typedef --- @field line integer -- Source line of the typedef
--- @field bytes integer -- total byte size --- @field bytes integer -- Total byte size
--- @field fields BindsField[] -- the field list --- @field fields BindsField[] -- The field list
--- @class MacroEntry --- @class MacroEntry
--- @field name string -- macro name (e.g. "WORD_COUNT(my_macro, 4)") --- @field name string -- Macro name (e.g. "WORD_COUNT(my_macro, 4)")
--- @field line integer -- source line --- @field line integer -- Source line
--- @field words integer -- declared word count --- @field words integer -- Declared word count
--- @class Finding --- @class Finding
--- @field line integer -- source line --- @field line integer -- Source line
--- @field msg string -- finding message --- @field msg string -- Finding message
--- @class AnnotationResult --- @class AnnotationResult
--- @field source string -- set by this pass; original source path --- @field source string -- Set by this pass; original source path
--- @field atoms AtomEntry[] -- atom declarations in this source --- @field atoms AtomEntry[] -- Atom declarations in this source
--- @field annots AnnotEntry[] -- annotation entries --- @field annots AnnotEntry[] -- Annotation entries
--- @field macros MacroEntry[] -- macro word-count declarations --- @field macros MacroEntry[] -- Macro word-count declarations
--- @field binds BindsStruct[] -- Binds_* struct declarations --- @field binds BindsStruct[] -- Binds_* struct declarations
--- @field errors Finding[] -- errors from validation --- @field errors Finding[] -- Errors from validation
--- @field warnings Finding[] -- warnings from validation --- @field warnings Finding[] -- Warnings from validation
--- @field info table -- info summary (not rendered here) --- @field info table -- Info summary (not rendered here)
--- @class ModuleEntry --- @class ModuleEntry
--- @field dir string -- absolute directory path --- @field dir string -- Absolute directory path
--- @field dir_basename string -- basename (e.g. "duffle", "gte_hello") --- @field dir_basename string -- Basename (e.g. "duffle", "gte_hello")
--- @field atoms_count integer -- pre-counted atoms for filtering --- @field atoms_count integer -- Pre-counted atoms for filtering
--- @class ModuleReport --- @class ModuleReport
--- @field dir string -- module directory --- @field dir string -- Module directory
--- @field sources SourceFile[] -- sources in this module --- @field sources SourceFile[] -- Sources in this module
--- @field results AnnotationResult[] -- per-source validate() results --- @field results AnnotationResult[] -- Per-source validate() results
--- @class ProjectReport --- @class ProjectReport
--- @field results AnnotationResult[] -- all per-source results --- @field results AnnotationResult[] -- All per-source results
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Per-MODULE annotation report (aggregated across all sources in a dir) -- Per-MODULE annotation report (aggregated across all sources in a dir)
@@ -148,328 +147,453 @@ local function source_basename(path)
return path:match(BASENAME_PATTERN) or path return path:match(BASENAME_PATTERN) or path
end end
--- (internal) Format a single annotation entry as one rendered line. -- ════════════════════════════════════════════════════════════════════════════
--- @param a AnnotEntry -- Markdown renderers (consolidated-report-files refactor, 2026-07-26)
--- @param src_name string -- ════════════════════════════════════════════════════════════════════════════
--- Render the thin project-wide summary (`build/atom_meta_report.summary.md`).
--- @param all_results {
--- module:string,
--- atoms:integer,
--- annots:integer,
--- binds:integer,
--- macros:integer,
--- findings:integer,
--- errors:integer,
--- warnings:integer,
--- info:integer }[]
--- @return string --- @return string
local function format_annot_line(a, src_name) local function render_project_summary(all_results)
if a.error then local lines = {
return string.format(" ✗ line %d %s [ERROR: %s] [%s]", a.line, a.macro or "?", a.error, src_name) "# Project summary",
end "> Auto-generated by ps1_meta.lua (passes/report.lua).",
local line = string.format(" ● line %d %s [%s]", a.line, a.name, src_name) "",
if a.binds then line = line .. " binds=" .. a.binds end "| module | atoms | annots | binds | macros | findings | errors | warnings | info |",
if #a.reads > 0 then line = line .. " reads={" .. table.concat(a.reads, ",") .. "}" end "|--------|-------|--------|-------|--------|----------|--------|----------|------|",
if #a.writes > 0 then line = line .. " writes={" .. table.concat(a.writes, ",") .. "}" end
return line
end
--- (internal) Tally totals across all results in a module.
--- @param results AnnotationResult[]
--- @return integer, integer, integer, integer, integer, integer
local function tally_module_totals(results)
local total_atoms, total_annots, total_binds, total_macros = 0, 0, 0, 0
local total_errors, total_warnings = 0, 0
for _, r in ipairs(results) do
total_atoms = total_atoms + #r.atoms
total_annots = total_annots + #r.annots
total_binds = total_binds + #r.binds
total_macros = total_macros + #r.macros
total_errors = total_errors + #r.errors
total_warnings = total_warnings + #r.warnings
end
return total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings
end
-- (internal) Section renderer: per-source atom declarations.
local function render_module_atoms_section(add, results)
add(SECTION_HEADER_ATOMS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do
add(string.format(" MipsAtom_(%s) line %d [%s]", a.name, a.line, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source annotation entries.
local function render_module_annots_section(add, results)
add(SECTION_HEADER_ANNOTS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
add(format_annot_line(a, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source Binds_* struct declarations.
local function render_module_binds_section(add, results)
add(SECTION_HEADER_BINDS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format(" %s line %d %d bytes [%s]", b.name, b.line, b.bytes, src_name))
for _, f in ipairs(b.fields) do
add(string.format(" +%2d: %s", f.offset, f.name))
end
end
end
add("")
end
-- (internal) Section renderer: per-source macro word-count declarations.
local function render_module_macros_section(add, results)
add(SECTION_HEADER_MACROS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format(" %s line %d words=%d [%s]", m.name, m.line, m.words, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source errors (one-line + "(none)" if empty).
local function render_module_errors_section(add, results, total_errors)
add(SECTION_HEADER_ERRORS)
if total_errors == 0 then
add(" (none)")
else
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, e in ipairs(r.errors) do
add(string.format(" ✗ line %d %s [%s]", e.line, e.msg, src_name))
end
end
end
add("")
end
-- (internal) Section renderer: per-source warnings (one-line + "(none)" if empty).
local function render_module_warnings_section(add, results, total_warnings)
add(SECTION_HEADER_WARNINGS)
if total_warnings == 0 then
add(" (none)")
else
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, w in ipairs(r.warnings) do
add(string.format(" ⚠ line %d %s [%s]", w.line, w.msg, src_name))
end
end
end
add("")
end
-- ════════════════════════════════════════════════════════════════════════════
-- SECTION_RENDERERS — data-driven section dispatch (the plex pattern)
-- ════════════════════════════════════════════════════════════════════════════
--
-- Each entry maps a section to its (header, render_fn). The render_fn signature:
-- render_fn(add, results, totals)
-- add -- the `add(line)` closure from the surrounding report renderer
-- results -- AnnotationResult[] (per-source results)
-- totals -- {atoms, annots, binds, macros, errors, warnings} counts
--
-- Sections that need to render "(none)" vs iterate use totals.errors / totals.warnings;
-- other sections ignore the totals arg.
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
local SECTION_RENDERERS = {
{ header = SECTION_HEADER_ATOMS, render = render_module_atoms_section },
{ header = SECTION_HEADER_ANNOTS, render = render_module_annots_section },
{ header = SECTION_HEADER_BINDS, render = render_module_binds_section },
{ header = SECTION_HEADER_MACROS, render = render_module_macros_section },
{ header = SECTION_HEADER_ERRORS, render = function(add, results, totals) return render_module_errors_section(add, results, totals.errors) end },
{ header = SECTION_HEADER_WARNINGS, render = function(add, results, totals) return render_module_warnings_section(add, results, totals.warnings) end },
}
--- Render the per-MODULE annotation report (one `<dir_basename>.annotations.txt`).
--- @param dir string -- module directory path
--- @param sources SourceFile[] -- sources in this module
--- @param results AnnotationResult[] -- per-source validate() results
--- @return string -- the rendered report text
local function render_module_report(dir, sources, results)
local lines = {}
local function add(s) lines[#lines + 1] = s end
add(RULE_THICK)
add("ANNOTATION PASS — module " .. source_basename(dir))
add(RULE_THICK)
add(string.format("Sources: %d", #sources))
for _, s in ipairs(sources) do add(" " .. s.path) end
add("")
local total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings = tally_module_totals(results)
add(string.format("Atoms: %d Annotations: %d Binds structs: %d Macro decls: %d",
total_atoms, total_annots, total_binds, total_macros))
add("")
-- Bundle the totals so the section renderers don't need separate parameter lists.
-- Errors/warnings sections need their total count to decide "(none)" vs iterate.
-- Sections without totals (atoms/annots/binds/macros) ignore this arg.
local totals = {
atoms = total_atoms, annots = total_annots, binds = total_binds,
macros = total_macros, errors = total_errors, warnings = total_warnings,
} }
local totals = { atoms = 0, annots = 0, binds = 0, macros = 0, findings = 0, errors = 0, warnings = 0, info = 0 }
-- THE per-section dispatch. ONE loop over SECTION_RENDERERS. for _, e in ipairs(all_results) do
-- Each renderer writes its header + content via the `add` closure (pre-bound above). lines[#lines + 1] = string.format("| %s | %d | %d | %d | %d | %d | %d | %d | %d |"
-- Adding a new section = 1 row here + 1 render_<thing>_section function. , e.module, e.atoms, e.annots, e.binds, e.macros, e.findings, e.errors, e.warnings, e.info)
for _, section in ipairs(SECTION_RENDERERS) do totals.atoms = totals.atoms + e.atoms
section.render(add, results, totals) totals.annots = totals.annots + e.annots
totals.binds = totals.binds + e.binds
totals.macros = totals.macros + e.macros
totals.findings = totals.findings + e.findings
totals.errors = totals.errors + e.errors
totals.warnings = totals.warnings + e.warnings
totals.info = totals.info + e.info
end end
lines[#lines + 1] = string.format("| **TOTAL** | %d | %d | %d | %d | %d | %d | %d | %d |"
, totals.atoms, totals.annots, totals.binds, totals.macros, totals.findings, totals.errors, totals.warnings, totals.info)
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
-- ════════════════════════════════════════════════════════════════════════════ --- Render the per-module verbose source-map markdown (`build/<module>.atoms.md`).
-- Per-project summary --- Per-source sub-section, per-atom stanza with sourcemap + provenance rows.
-- ════════════════════════════════════════════════════════════════════════════ --- Pulls sourcemap + provenance from `atoms_source_map` (no second source walk).
--- @param dir string
--- Render the per-project summary (`build/gen/annotation_validation.txt`). --- @param dir_sources SourceFile[]
--- Aggregates totals across all sources; lists per-source error counts if any source has errors. --- @param wc table<string, integer>
--- @param all_results AnnotationResult[]
--- @return string --- @return string
local function render_project_report(all_results) local function render_module_atoms_md(dir, dir_sources, wc)
local lines = {} local dir_basename = source_basename(dir)
local lines = {
"# " .. dir_basename .. " — atoms (verbose source map)",
"> Per-word call-site + provenance. Auto-generated.",
"",
}
for _, src in ipairs(dir_sources) do
local src_name = source_basename(src.path)
lines[#lines + 1] = "## " .. src_name
lines[#lines + 1] = ""
-- For each atom with a projection, render its sourcemap + provenance.
local atoms_list = {}
for _, atom in ipairs((src.scan or {}).atoms or {}) do
if atom.paths then atoms_list[#atoms_list + 1] = atom end
end
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
if atom.paths then atoms_list[#atoms_list + 1] = atom end
end
if #atoms_list == 0 then
lines[#lines + 1] = "_(no atom projections)_"
lines[#lines + 1] = ""
else
-- Per-source forward-slash path (same one `emit_atom_stanza` / `emit_provenance_stanza` would derive;
-- computed once per `## <source>` heading and reused by each atom's `WORD N CALL ...` field).
local rel_path = src.path:gsub("\\\\", "/")
for _, atom in ipairs(atoms_list) do
lines[#lines + 1] = string.format(
"### atom: %s (line %d, %d words)",
atom.name, atom.line or 0, #(atom.paths.items or {}))
lines[#lines + 1] = ""
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
lines[#lines + 1] = "```"
-- Per-atom invariant: call the per-atom renderers, NOT the per-source ones.
-- The per-source renderers enumerate every atom in `src`;
-- calling them in a per-atom loop would repeat the whole source under every `### atom:` heading.
lines[#lines + 1] = atoms_source_map.render_atom_source_map(atom):gsub("\n+$", "")
lines[#lines + 1] = "```"
lines[#lines + 1] = ""
lines[#lines + 1] = "**Provenance** — per-word definition + body:"
lines[#lines + 1] = "```"
lines[#lines + 1] = atoms_source_map.render_atom_provenance(atom, wc, rel_path):gsub("\n+$", "")
lines[#lines + 1] = "```"
lines[#lines + 1] = ""
end
end
end
return table.concat(lines, "\n") .. "\n"
end
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
--- Aggregates annotation + static-analysis content across all sources in `dir`.
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
--- @param dir string
--- @param dir_sources SourceFile[]
--- @param annot_results AnnotationResult[]
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
--- @return string
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
local dir_basename = source_basename(dir)
local lines = {
"# " .. dir_basename .. " — atom meta report",
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
"",
}
local function add(s) lines[#lines + 1] = s end local function add(s) lines[#lines + 1] = s end
local total_atoms, total_annots, total_macros, total_binds = 0, 0, 0, 0 -- Module summary table.
local total_errors, total_warnings = 0, 0 local n_atoms = 0
for _, r in ipairs(all_results) do local n_annot = 0
total_atoms = total_atoms + #r.atoms local n_binds = 0
total_annots = total_annots + #r.annots local n_macros = 0
total_macros = total_macros + #r.macros local n_bare, n_proc = 0, 0
total_binds = total_binds + #r.binds for _, r in ipairs(annot_results) do
total_errors = total_errors + #r.errors n_atoms = n_atoms + #r.atoms
total_warnings = total_warnings + #r.warnings n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
for _, a in ipairs(sa_results.atoms or {}) do
if a.kind == "comp_bare" then n_bare = n_bare + 1
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
end
end end
add(RULE_THICK) add("## Module summary"); add("")
add("ANNOTATION VALIDATION — project summary") add("| metric | value |"); add("|--------|-------|")
add(RULE_THICK) add(string.format("| sources | %d |", #dir_sources))
add("") add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
add(string.format("Atoms: %d", total_atoms)) #(sa_results.atoms or {}),
add(string.format("Annotations: %d", total_annots)) #(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
add(string.format("Macros: %d", total_macros)) add(string.format("| annotations | %d |", n_annot))
add(string.format("Binds: %d", total_binds)) add(string.format("| binds structs | %d |", n_binds))
add("") add(string.format("| macro decls | %d |", n_macros))
add(string.format("Errors: %d", total_errors)) add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
add(string.format("Warnings: %d", total_warnings)) #(sa_results.findings or {}),
#(sa_results.errors or {}),
#(sa_results.warnings or {}),
#(sa_results.info or {})))
add("") add("")
if total_errors > 0 then -- Sources
add("Per-source error counts:") add("## Sources"); add("")
for _, r in ipairs(all_results) do for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
if #r.errors > 0 then add("")
local src_name = source_basename(r.source)
add(string.format(" %s : %d error(s)", src_name, #r.errors)) -- Atoms (annotation)
add("## Atoms"); add("")
add("| kind | name | source | line |"); add("|------|------|--------|------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
end
end
add("")
-- Annotations
add("## Annotations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | name | binds | reads | writes |")
add("|--------|------|------|-------|-------|--------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
local binds = a.binds or ""
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or ""
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or ""
add(string.format("| %s | %d | %s | %s | %s | %s |"
, src_name, a.line, a.name, binds, reads, writes))
end
end
end
add("")
-- Binds_* structs
add("## Binds_* structs"); add("")
if #annot_results == 0 then
add("_(none)_")
else
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format("### %s (%s:%d, %d bytes)",
b.name, src_name, b.line, b.bytes))
for _, f in ipairs(b.fields) do
add(string.format("- `+%d %s`", f.offset, f.name))
end
add("")
end
end
end
-- Macro decls
add("## Macro word-count declarations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | macro declaration |")
add("|--------|------|-------------------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format("| %s | %d | %s |",
src_name, m.line, m.name))
end
end
end
add("")
-- Findings by atom (static-analysis)
add("## Static analysis — findings by atom"); add("")
local by_atom = {}
for _, f in ipairs(sa_results.findings or {}) do
by_atom[f.atom] = by_atom[f.atom] or {}
by_atom[f.atom][#by_atom[f.atom] + 1] = f
end
if next(by_atom) == nil then
add("_(no findings)_")
else
for _, a in ipairs(sa_results.atoms or {}) do
local fs = by_atom[a.name]
if fs then
add(string.format("### %s", a.name))
for _, f in ipairs(fs) do
add(string.format("- `[%s] %s`", f.check, f.msg))
end
add("")
end
end
end
-- Errors / Warnings / Info
local function add_findings(label, entries)
add(string.format("## %s", label))
if #entries == 0 then
add("_(none)_")
else
for _, e in ipairs(entries) do
add(string.format("- line %d %s", e.line, e.msg))
end end
end end
add("") add("")
end end
add_findings("Errors", sa_results.errors or {})
add_findings("Warnings", sa_results.warnings or {})
add_findings("Info", sa_results.info or {})
-- Per-atom cycle counts (path-aware)
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
add("| atom | source | min | max | branches | paths | notes |")
add("|------|--------|-----|-----|----------|-------|-------|")
local sorted = {}
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
table.sort(sorted, function(x, y)
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
end)
for _, a in ipairs(sorted) do
local p = a.paths or {}
local src_name = a.source_path and source_basename(a.source_path) or ""
local notes = ""
if p.has_loops then notes = notes .. " [loop!]" end
if p.unknown_macros and #p.unknown_macros > 0 then
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
end
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
a.name, src_name,
p.cycles_min or 0, p.cycles_max or 0,
p.branches or 0, p.paths or 0, notes))
end
add("")
-- Per-source scan summary
add("## Per-source scan summary"); add("")
for _, src in ipairs(dir_sources) do
local src_atoms = {}
for _, a in ipairs(sa_results.atoms or {}) do
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
end
if #src_atoms > 0 then
local mn, mx = math.huge, -1
for _, a in ipairs(src_atoms) do
local p = a.paths or {}
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
end
local path_str
if mx > 0 then
path_str = string.format(" cycles=%d..%d", mn, mx)
else
path_str = string.format(" %d cycles", mn)
end
add(string.format("- `%s` — %d atom%s%s",
src.basename, #src_atoms,
#src_atoms == 1 and "" or "s", path_str))
end
end
add("")
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Orchestration helpers -- REPORT_RENDERERS — data-driven report dispatch (one row per file kind)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- `once = true` means render once at the project level (not per-module).
--- (internal) Re-validate every source in a directory against the canonical corpus projection. -- `basename(dir_basename)` yields the file's basename for that kind.
--- Calls `annotation.validate()` per source to produce the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings) -- `gather(ctx, dir, dir_sources [, all_modules])` returns the rendered string.
--- that the report renderer consumes. Eeach report pass run is reproducible from the corpus. local REPORT_RENDERERS = {
--- Returns the list of module results + the flat list of all results (for the project-wide summary). {
--- @param ctx PassCtx name = "atom_meta_report",
--- @param dir_sources SourceFile[] ext = "md",
--- @return AnnotationResult[], AnnotationResult[] basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
local function lookup_module_results(ctx, dir_sources) once = false,
local module_results = {} gather = function(ctx, dir, dir_sources)
local all_results = {} -- Annotations: re-run `annotation.validate()` per source (the existing pattern).
for _, src in ipairs(dir_sources) do local annot_results = {}
if src.scan then for _, src in ipairs(dir_sources) do
local result = annotation.validate(ctx, src, nil) if src.scan then
result.source = src.path -- tag for downstream rendering local r = annotation.validate(ctx, src, nil)
module_results[#module_results + 1] = result r.source = src.path
all_results[#all_results + 1] = result annot_results[#annot_results + 1] = r
end end
end end
return module_results, all_results -- Static-analysis: read stashed projection (no re-validate).
end local dir_basename = dir:match("([^/\\]+)$") or dir
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
--- (internal) Does this module's results contain anything worth emitting? return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
--- @param module_results AnnotationResult[] end,
--- @return boolean },
local function module_has_content(module_results) {
for _, r in ipairs(module_results) do name = "atoms",
if #r.atoms > 0 or #r.annots > 0 or #r.binds > 0 ext = "md",
or #r.macros > 0 or #r.errors > 0 or #r.warnings > 0 then basename = function(dir_basename) return dir_basename .. ".atoms" end,
return true once = false,
end gather = function(ctx, dir, dir_sources)
end return render_module_atoms_md(dir, dir_sources,
return false ctx.shared.corpus.word_counts or {})
end end,
},
--- (internal) Log a debug message if `_G[DEBUG_FLAG]` is truthy. {
--- @param fmt string name = "summary",
local function debug_log(fmt, ...) ext = "md",
if _G[DEBUG_FLAG] then basename = function(_dir_basename) return "atom_meta_report.summary" end,
io.stderr:write(string.format("[%s] " .. fmt, PASS_NAME, ...)) once = true,
end gather = function(_ctx, _dir, _dir_sources, all_modules)
end return render_project_summary(all_modules)
end,
},
}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M — module exports -- M — public pass surface
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
local M = {} local M = {}
--- Run the report pass. --- Run the report pass. Emits 1 `atom_meta_report.summary.md` per build + 2 `atom_meta_report.md` + 2 `atoms.md` files per module (duffle + gte_hello).
--- Renders one `<dir_basename>.annotations.txt` per source-directory that has content, plus the project-wide `annotation_validation.txt` summary. --- Reads `corpus.static_analysis_results` (added in Phase 1) to populate per-module findings without re-running validate().
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
local outputs = {} local outputs = {}
local errors = {} local corpus = ctx.shared and ctx.shared.corpus
local warnings = {} local by_dir = (corpus and corpus.sources_by_dir) or {}
-- Module grouping comes from `corpus.sources_by_dir` (the canonical projection). -- `out_path_root`: when the conventional `out_root` is `build/gen` (any spelling — relative, absolute, separator variants).
-- Iterate it directly; no private cache, no per-pass stash. -- Write the md files to `build/` (parent of `gen/`) instead of nested under `gen/`.
local corpus = ctx.shared and ctx.shared.corpus -- Mirrors the `gdb_tape_atoms_runtime.gdb` relocation.
local by_dir = (corpus and corpus.sources_by_dir) or {} local function ends_with_gen(p)
return type(p) == "string" and (p:match("[/\\]gen[/\\]?$") ~= nil
or p == "build/gen" or p == "build\\gen")
end
local out_root_effective = ends_with_gen(ctx.out_root)
and ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
or ctx.out_root
duffle.ensure_dir(ctx.out_root) duffle.ensure_dir(out_root_effective)
-- Aggregator for the project-wide `once = true` summary renderer.
local all_modules = {}
local all_results_for_summary = {}
for dir, dir_sources in pairs(by_dir) do for dir, dir_sources in pairs(by_dir) do
local dir_basename = dir:match("([^/\\]+)$") or dir local dir_basename = dir:match("([^/\\]+)$") or dir
debug_log("dir=%s basename=%s sources=%d\n", dir, dir_basename, #dir_sources)
if #dir_sources > 0 then -- Per-renderer dispatch for the per-module renderers (once = false).
local module_results, all_results = lookup_module_results(ctx, dir_sources) for _, renderer in ipairs(REPORT_RENDERERS) do
for _, r in ipairs(all_results) do if not renderer.once then
all_results_for_summary[#all_results_for_summary + 1] = r local body = renderer.gather(ctx, dir, dir_sources)
local out_path = out_root_effective .. "/" .. renderer.basename(dir_basename) .. "." .. renderer.ext
duffle.write_file(out_path, body)
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
end end
end
if module_has_content(module_results) then -- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
local out_path = ctx.out_root .. "/" .. dir_basename .. ".annotations.txt" local annot_results = {}
duffle.write_file(out_path, render_module_report(dir, dir_sources, module_results)) for _, src in ipairs(dir_sources) do
outputs[#outputs + 1] = { annotations_txt = out_path } if src.scan then
else local r = annotation.validate(ctx, src, nil)
debug_log(" -> no content; skipping\n") r.source = src.path
annot_results[#annot_results + 1] = r
end end
end end
local n_annot, n_binds, n_macros = 0, 0, 0
for _, r in ipairs(annot_results) do
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
all_modules[#all_modules + 1] = {
module = dir_basename,
atoms = #(sa_results.atoms or {}),
annots = n_annot,
binds = n_binds,
macros = n_macros,
findings = #(sa_results.findings or {}),
errors = #(sa_results.errors or {}),
warnings = #(sa_results.warnings or {}),
info = #(sa_results.info or {}),
}
end
-- Project-wide renderer (once = true): write the summary file.
for _, renderer in ipairs(REPORT_RENDERERS) do
if renderer.once then
local body = renderer.gather(ctx, nil, nil, all_modules)
local out_path = out_root_effective .. "/" .. renderer.basename("") .. "." .. renderer.ext
duffle.write_file(out_path, body)
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
end
end end
if #all_results_for_summary > 0 then return { outputs = outputs, errors = {}, warnings = {} }
local summary_path = ctx.out_root .. "/annotation_validation.txt"
duffle.write_file(summary_path, render_project_report(all_results_for_summary))
outputs[#outputs + 1] = { summary_txt = summary_path }
end
return { outputs = outputs, errors = errors, warnings = warnings }
end end
return M return M
+358 -166
View File
@@ -2,8 +2,8 @@
--- ---
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once, --- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once,
--- extracting every construct type the metaprograms need: --- extracting every construct type the metaprograms need:
---
--- MipsAtom_ (kind = "atom", with optional atom_info inner) --- MipsAtom_ (kind = "atom", with optional atom_info inner)
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
--- MipsAtomComp_ (kind = "comp_bare") --- MipsAtomComp_ (kind = "comp_bare")
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {}) --- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates --- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
@@ -14,8 +14,6 @@
--- The result is attached to each `src.scan` so downstream passes can read from `src.scan.atoms` / `src.scan.binds` / etc. without re-walking the source. --- The result is attached to each `src.scan` so downstream passes can read from `src.scan.atoms` / `src.scan.binds` / etc. without re-walking the source.
--- This is the first pass in the dep graph (no deps). --- This is the first pass in the dep graph (no deps).
--- Every other pass that reads source structure depends on this one — see `ps1_meta.lua :: PASSES`. --- Every other pass that reads source structure depends on this one — see `ps1_meta.lua :: PASSES`.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale. -- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
@@ -23,7 +21,7 @@
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when required). -- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when required).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Forward declarations for helpers used by earlier parsers (parse_enum_body_fields needs parse_enum_int_literal; -- Forward declarations for helpers used by earlier parsers (parse_enum_body_fields needs parse_enum_int_literal;
-- parse_typedef_binds needs duffle.find_byte). -- parse_typedef_binds needs duffle.find_byte).
@@ -37,7 +35,7 @@ local parse_enum_int_literal
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class SourceScan --- @class SourceScan
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtomComp_ + MipsAtomComp_Proc_ --- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtom_Proc_ + MipsAtomComp_ + MipsAtomComp_Proc_
--- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only) --- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only)
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed) --- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed) --- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
@@ -50,15 +48,15 @@ local parse_enum_int_literal
--- @field line_of fun(pos: integer): integer -- shared LineIndex closure --- @field line_of fun(pos: integer): integer -- shared LineIndex closure
--- @class DebugSkipMarker --- @class DebugSkipMarker
--- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive; any other ident reaches the unrelated fallback and is never associated with a declaration. --- @field marker_kind string -- Exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive; any other ident reaches the unrelated fallback and is never associated with a declaration.
--- @field marker_line integer -- line of the marker ident start --- @field marker_line integer -- Line of the marker ident start
--- @field marker_pos integer -- byte position of the marker ident start (the comment walker anchors here) --- @field marker_pos integer -- Byte position of the marker ident start (the comment walker anchors here)
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form) --- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
--- @field has_parens boolean -- true iff a `(...)` follows the marker ident (diagnostic-only) --- @field has_parens boolean -- true iff a `(...)` follows the marker ident (diagnostic-only)
--- @field args string|nil -- trimmed args inside the `(...)` (nil when has_parens is false) --- @field args string|nil -- Trimmed args inside the `(...)` (nil when has_parens is false)
--- @field pending boolean -- true while awaiting the following declaration --- @field pending boolean -- true while awaiting the following declaration
--- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot --- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed) --- @field target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
--- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_` --- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_`
--- @class RegTypeDefault --- @class RegTypeDefault
@@ -71,15 +69,15 @@ local parse_enum_int_literal
--- @field reg string -- "R_T0" --- @field reg string -- "R_T0"
--- @field type_name string --- @field type_name string
--- @field pointer_depth integer --- @field pointer_depth integer
--- @field source_line integer -- line of the call site (callsite or enum-site) --- @field source_line integer -- Line of the call site (callsite or enum-site)
--- @class AtomCtxEntry --- @class AtomCtxEntry
--- @field rbind_atom string -- the rbind atom ident that this consumer should propagate types from --- @field rbind_atom string -- The rbind atom ident that this consumer should propagate types from
--- @field info_line integer --- @field info_line integer
--- @field source string -- absolute path of the source file --- @field source string -- Absolute path of the source file
--- @class AtomPhaseGroup --- @class AtomPhaseGroup
--- @field atoms string[] -- atom names tagged with this phase label (source-order) --- @field atoms string[] -- Atom names tagged with this phase label (source-order)
--- @class AtomViewEntry --- @class AtomViewEntry
--- @field atom_name string -- e.g. "red_cube_g4_face" --- @field atom_name string -- e.g. "red_cube_g4_face"
@@ -88,11 +86,11 @@ local parse_enum_int_literal
--- @field info_line integer -- line of the atom_info call --- @field info_line integer -- line of the atom_info call
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- Absolute path to the source file
--- @field text string -- the full source text --- @field text string -- Full source text
--- @field dir string -- the directory containing the source --- @field dir string -- Directory containing the source
--- @field basename string -- filename without extension --- @field basename string -- Filename without extension
--- @field scan table -- pre-scanned SourceScan payload (set by this pass) --- @field scan table -- Pre-scanned SourceScan payload (set by this pass)
--- @class PassCtx --- @class PassCtx
--- @field sources SourceFile[] --- @field sources SourceFile[]
@@ -111,15 +109,15 @@ local parse_enum_int_literal
--- @class AtomEntry --- @class AtomEntry
--- @field line integer --- @field line integer
--- @field name string -- atom name (for components: without ac_ prefix) --- @field name string -- Atom name (for components: without ac_ prefix)
--- @field body string -- brace-delimited body (without the braces) --- @field body string -- Brace-delimited body (without the braces)
--- @field body_off integer -- char offset of body[1] in source --- @field body_off integer -- Char offset of body[1] in source
--- @field kind string -- "atom" | "comp_bare" | "comp_proc" | "raw_atom" --- @field kind string -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "raw_atom"
--- @field raw_name string -- un-stripped name (for components: with ac_ prefix) --- @field raw_name string -- Un-stripped name (for components: with ac_ prefix)
--- @field ident_pos integer -- position of the MipsAtom_/MipsAtomComp_ ident start --- @field ident_pos integer -- Position of the MipsAtom_/MipsAtomComp_ ident start
--- @field after_paren integer -- position past the closing paren --- @field after_paren integer -- Position past the closing paren
--- @field debug_skip boolean -- true when an `atom_dbg_skip` bare marker immediately precedes this declaration (sole-owner stamp; see push_debug_skip_marker) --- @field debug_skip boolean -- true when an `atom_dbg_skip` bare marker immediately precedes this declaration (sole-owner stamp; see push_debug_skip_marker)
--- @field declaration_comment string|nil -- populated by the scanner (backward walk past the marker, captures contiguous `/* */` or `//` block) --- @field declaration_comment string|nil -- Populated by the scanner (backward walk past the marker, captures contiguous `/* */` or `//` block)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Local helpers (shared by per-form parsers) -- Local helpers (shared by per-form parsers)
@@ -139,10 +137,10 @@ local QUALIFIER_KEYWORDS = {
local AC_PREFIX = "ac_" local AC_PREFIX = "ac_"
local AC_PREFIX_LEN = 3 local AC_PREFIX_LEN = 3
-- Strip the "ac_" prefix from a component name. --- Strip the "ac_" prefix from a component name.
-- Returns the input unchanged if it doesn't start with the prefix. --- Returns the input unchanged if it doesn't start with the prefix.
-- @param raw_name string --- @param raw_name string
-- @return string --- @return string
local function strip_ac_prefix(raw_name) local function strip_ac_prefix(raw_name)
if #raw_name > AC_PREFIX_LEN and raw_name:sub(1, AC_PREFIX_LEN) == AC_PREFIX then if #raw_name > AC_PREFIX_LEN and raw_name:sub(1, AC_PREFIX_LEN) == AC_PREFIX then
return raw_name:sub(AC_PREFIX_LEN + 1) return raw_name:sub(AC_PREFIX_LEN + 1)
@@ -156,7 +154,7 @@ end
local function push_debug_skip_marker(out, marker) local function push_debug_skip_marker(out, marker)
local markers = out.debug_skip_markers local markers = out.debug_skip_markers
local prior = markers[#markers] local prior = markers[#markers]
if prior and prior.pending then if prior and prior.pending then
prior.pending = false prior.pending = false
prior.superseded_by_marker_line = marker.marker_line prior.superseded_by_marker_line = marker.marker_line
end end
@@ -178,25 +176,25 @@ end
-- Returns (body, after_brace, body_off) on success, or (nil, fallback_pos) on no brace. -- Returns (body, after_brace, body_off) on success, or (nil, fallback_pos) on no brace.
-- `fallback_pos` defaults to `after_paren + 1` (the common "advance by 1" case). -- `fallback_pos` defaults to `after_paren + 1` (the common "advance by 1" case).
local function find_body_braces(source, after_paren, fallback) local function find_body_braces(source, after_paren, fallback)
local brace = duffle.scan_to_char(source, "{", after_paren) local brace = duffle.scan_to_char(source, "{", after_paren)
if not brace then return nil, fallback or (after_paren + 1) end if not brace then return nil, fallback or (after_paren + 1) end
local body, after_brace = duffle.read_braces(source, brace) local body, after_brace = duffle.read_braces(source, brace)
return body, after_brace, brace + 1 return body, after_brace, brace + 1
end end
-- Walk backward from `start_pos` capturing contiguous `/* */` block(s) and --- Walk backward from `start_pos` capturing contiguous `/* */` block(s) and
-- `//` line(s) that immediately precede it. The caller (preceding_declaration_comment) --- `//` line(s) that immediately precede it. The caller (preceding_declaration_comment)
-- supplies `start_pos` so the walker does not need to detect marker shape or prelude layout. --- supplies `start_pos` so the walker does not need to detect marker shape or prelude layout.
-- The scanner already knows the marker_pos + decl ident_pos and threads that knowledge forward. --- The scanner already knows the marker_pos + decl ident_pos and threads that knowledge forward.
-- ---
-- The walker captures: --- The walker captures:
-- - Block comment close `*/` followed by walking back to `/*`. --- - Block comment close `*/` followed by walking back to `/*`.
-- - `//` line comments (the line containing the current non-ws position starts with `//`). --- - `//` line comments (the line containing the current non-ws position starts with `//`).
-- It stops at the first non-ws char that does not begin a comment block or line. --- It stops at the first non-ws char that does not begin a comment block or line.
-- Empty string if no comment is adjacent. --- Empty string if no comment is adjacent.
-- @param source string --- @param source string
-- @param start_pos integer -- exclusive upper bound for the captured block --- @param start_pos integer -- exclusive upper bound for the captured block
-- @return string --- @return string
local function preceding_comment_walk_backward(source, start_pos) local function preceding_comment_walk_backward(source, start_pos)
local pieces = {} local pieces = {}
local scan_pos = start_pos local scan_pos = start_pos
@@ -249,13 +247,13 @@ local function preceding_comment_walk_backward(source, start_pos)
return table.concat(pieces, "\n") return table.concat(pieces, "\n")
end end
-- Resolve the start position for the declaration-comment walk. --- Resolve the start position for the declaration-comment walk.
-- When a debug-skip marker is pending, the walker must start from the position immediately before the marker ident --- When a debug-skip marker is pending, the walker must start from the position immediately before the marker ident
-- (so it walks backward past the marker text and any `FI_ MipsAtom ac_X(args)` proc-prelude layout — neither of which is visible if we start from the declaration ident_pos). --- (so it walks backward past the marker text and any `FI_ MipsAtom ac_X(args)` proc-prelude layout — neither of which is visible if we start from the declaration ident_pos).
-- When no marker is pending, the walker starts from the declaration ident_pos directly. --- When no marker is pending, the walker starts from the declaration ident_pos directly.
-- @param pending_marker DebugSkipMarker|nil --- @param pending_marker DebugSkipMarker|nil
-- @param ident_pos integer -- declaration ident position --- @param ident_pos integer -- declaration ident position
-- @return integer --- @return integer
local function comment_walk_start(pending_marker, ident_pos) local function comment_walk_start(pending_marker, ident_pos)
if pending_marker then if pending_marker then
return pending_marker.marker_pos - 1 return pending_marker.marker_pos - 1
@@ -263,16 +261,16 @@ local function comment_walk_start(pending_marker, ident_pos)
return ident_pos - 1 return ident_pos - 1
end end
-- Attach the pending marker to the next declaration. --- Attach the pending marker to the next declaration.
-- The declaration form disambiguates whole atoms from components; the resolved `debug_skip` is stamped directly on the declaration record --- The declaration form disambiguates whole atoms from components; the resolved `debug_skip` is stamped directly on the declaration record
-- (sole-owner discipline; see push_debug_skip_marker). --- (sole-owner discipline; see push_debug_skip_marker).
-- ---
-- A marker is POSITIVE (stamps `debug_skip = true` on the declaration) iff: --- A marker is POSITIVE (stamps `debug_skip = true` on the declaration) iff:
-- marker_kind == "atom_dbg_skip" AND is_bare == true --- marker_kind == "atom_dbg_skip" AND is_bare == true
-- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`. --- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`.
-- @param out SourceScan --- @param out SourceScan
-- @param target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed --- @param target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed
-- @return boolean|nil -- true iff the marker is the positive bare form --- @return boolean|nil -- true iff the marker is the positive bare form
local function attach_debug_skip_marker(out, target_kind) local function attach_debug_skip_marker(out, target_kind)
local markers = out.debug_skip_markers local markers = out.debug_skip_markers
local marker = markers[#markers] local marker = markers[#markers]
@@ -390,13 +388,13 @@ local function walk_body_fields(body, build_field)
while body_pos <= body_len do while body_pos <= body_len do
body_pos = duffle.skip_ws_and_cmt(body, body_pos) body_pos = duffle.skip_ws_and_cmt(body, body_pos)
if body_pos > body_len then break end if body_pos > body_len then break end
local first, first_end = duffle.read_ident(body, body_pos) local first, first_end = duffle.read_ident(body, body_pos)
if not first then if not first then
body_pos = body_pos + 1 body_pos = body_pos + 1
else else
local after_first = duffle.skip_ws_and_cmt(body, first_end) local after_first = duffle.skip_ws_and_cmt(body, first_end)
local result, new_pos = build_field(first, first_end, after_first) local result, new_pos = build_field(first, first_end, after_first)
if result then fields[#fields + 1] = result end if result then fields[#fields + 1] = result end
body_pos = new_pos or first_end body_pos = new_pos or first_end
-- Skip a single trailing `,` or `;`. -- Skip a single trailing `,` or `;`.
if body_pos <= body_len and (body:sub(body_pos, body_pos) == "," or body:sub(body_pos, body_pos) == ";") then if body_pos <= body_len and (body:sub(body_pos, body_pos) == "," or body:sub(body_pos, body_pos) == ";") then
@@ -442,7 +440,7 @@ local function parse_enum_body_fields(body)
local value local value
local new_pos local new_pos
if body:sub(after_name, after_name) == "=" then if body:sub(after_name, after_name) == "=" then
local val_pos = duffle.skip_ws_and_cmt(body, after_name + 1) local val_pos = duffle.skip_ws_and_cmt(body, after_name + 1)
local v, end_pos = parse_enum_int_literal(body, val_pos) local v, end_pos = parse_enum_int_literal(body, val_pos)
if v ~= nil then if v ~= nil then
value = v value = v
@@ -461,16 +459,16 @@ end
-- Returns a positive integer byte_size when the chain bottoms out at a builtin, or nil if the chain is broken, exceeds TYPE_CHAIN_MAX_DEPTH, or contains a cycle. -- Returns a positive integer byte_size when the chain bottoms out at a builtin, or nil if the chain is broken, exceeds TYPE_CHAIN_MAX_DEPTH, or contains a cycle.
local function resolve_typedef_byte_size(type_name, type_name_registry, visited, depth) local function resolve_typedef_byte_size(type_name, type_name_registry, visited, depth)
if depth > TYPE_CHAIN_MAX_DEPTH then return nil end if depth > TYPE_CHAIN_MAX_DEPTH then return nil end
if visited[type_name] then return nil end if visited[type_name] then return nil end
visited[type_name] = true visited[type_name] = true
-- Check the builtin primitive map FIRST. -- Check the builtin primitive map FIRST.
-- This handles undeclared builtin idents (e.g. `__UINT32_TYPE__` appears as underlying_type in `typedef __UINT32_TYPE__ TSet_(V4_S2);` -- This handles undeclared builtin idents (e.g. `__UINT32_TYPE__` appears as underlying_type in `typedef __UINT32_TYPE__ TSet_(V4_S2);`
-- even though the fixture never declares `__UINT32_TYPE__` itself). -- even though the fixture never declares `__UINT32_TYPE__` itself).
local builtin = BUILTIN_BYTE_SIZES[type_name] local builtin = BUILTIN_BYTE_SIZES[type_name]
if builtin ~= nil then return builtin end if builtin ~= nil then return builtin end
local entry = type_name_registry[type_name] local entry = type_name_registry[type_name]
if not entry then return nil end if not entry then return nil end
-- Confident: this entry was already resolved by the propagation pass (e.g., a builtin or a struct whose fields are all resolved). -- Confident: this entry was already resolved by the propagation pass (e.g., a builtin or a struct whose fields are all resolved).
@@ -519,9 +517,9 @@ local function propagate_type_sizes(out)
for name, entry in pairs(reg) do for name, entry in pairs(reg) do
if entry.byte_size == nil then if entry.byte_size == nil then
local resolved = resolve_typedef_byte_size(name, reg, {}, 1) local resolved = resolve_typedef_byte_size(name, reg, {}, 1)
if resolved ~= nil then if resolved ~= nil then
entry.byte_size = resolved entry.byte_size = resolved
any_change = true any_change = true
end end
end end
end end
@@ -802,6 +800,11 @@ local BYTE_x = 0x78 -- 'x'
local BYTE_X = 0x58 -- 'X' local BYTE_X = 0x58 -- 'X'
local BYTE_OPEN_BRACE = 0x7B -- '{' local BYTE_OPEN_BRACE = 0x7B -- '{'
local BYTE_CLOSE_BRACE= 0x7D -- '}' local BYTE_CLOSE_BRACE= 0x7D -- '}'
local BYTE_SLASH = 0x2F -- '/'
local BYTE_STAR = 0x2A -- '*'
local BYTE_SPACE = 0x20 -- ' '
local BYTE_TAB = 0x09 -- '\t'
local BYTE_CR = 0x0D -- '\r'
-- Maximum chain depth when resolving `R_*_Code` symbol RHS references. -- Maximum chain depth when resolving `R_*_Code` symbol RHS references.
-- Eight hops is enough for any production chain (R_TapePtr_Code -> R_T8_Code -> ...). -- Eight hops is enough for any production chain (R_TapePtr_Code -> R_T8_Code -> ...).
@@ -825,10 +828,48 @@ local function hex_digit_value(b)
return nil return nil
end end
-- Read one trailing C-comment that appears immediately after `pos` in `body`,
-- skipping horizontal whitespace and newlines first. Used by `parse_enum_entry` to
-- recover the `atom_auto_reg:` / `phase_auto_reg:` scope annotation embedded by
-- the `atom_auto_reg` / `phase_auto_reg` macros' RHS expansion
-- (`R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`).
-- Handles both block (`/* ... */`) and line (`// ...`) forms.
-- Returns the comment text (without delimiters), or nil if no comment is adjacent.
local function read_trailing_cmt_after(body, pos)
local body_len = #body
while pos <= body_len do
local b = body:byte(pos)
if b == BYTE_SPACE or b == BYTE_TAB or b == BYTE_NEWLINE or b == BYTE_CR then
pos = pos + 1
elseif b == BYTE_SLASH then
local b2 = body:byte(pos + 1)
if b2 == BYTE_STAR then
-- Block comment /* ... */
local i = pos + 2
while i < body_len do
if body:byte(i) == BYTE_STAR and body:byte(i + 1) == BYTE_SLASH then
return body:sub(pos + 2, i - 1)
end
i = i + 1
end
return nil -- unterminated; treat as no comment
elseif b2 == BYTE_SLASH then
-- Line comment // ... (strip the trailing newline)
local end_pos = duffle.find_byte(body, BYTE_NEWLINE, pos + 2) or (body_len + 1)
return body:sub(pos + 2, end_pos - 1)
end
return nil
else
return nil
end
end
return nil
end
--- Parse a decimal/negative-decimal/hex integer literal starting at byte position `start`. --- Parse a decimal/negative-decimal/hex integer literal starting at byte position `start`.
--- Returns (value, end_pos) on success, or (nil, start) on failure / no match. --- Returns (value, end_pos) on success, or (nil, start) on failure / no match.
--- Accepts: 12, -1, 0, 0x10, 0X1F, -0x10. --- Accepts: 12, -1, 0, 0x10, 0X1F, -0x10.
--- @param text string --- @param text string
--- @param start integer --- @param start integer
--- @return integer|nil, integer --- @return integer|nil, integer
--- Implementation note: this is a plain assignment (not `local function`) --- Implementation note: this is a plain assignment (not `local function`)
@@ -952,9 +993,9 @@ end
--- Always saves the raw RHS text into `code_macro_bodies` (for cross-source fallback during chain resolution), --- Always saves the raw RHS text into `code_macro_bodies` (for cross-source fallback during chain resolution),
--- then (if resolvable) stores the resolved integer code into `code_macros` keyed by the macro name. --- then (if resolvable) stores the resolved integer code into `code_macros` keyed by the macro name.
--- `directive_start` points at the `#` byte. The function is silent on non-matching directives, the caller skips the line in any case. --- `directive_start` points at the `#` byte. The function is silent on non-matching directives, the caller skips the line in any case.
--- @param source string --- @param source string
--- @param directive_start integer -- byte position of `#` --- @param directive_start integer -- byte position of `#`
--- @param code_macros table -- out._code_macros / ctx.shared._code_macros --- @param code_macros table -- out._code_macros / ctx.shared._code_macros
--- @param code_macro_bodies table -- out._code_macro_bodies / ctx.shared._code_macro_bodies --- @param code_macro_bodies table -- out._code_macro_bodies / ctx.shared._code_macro_bodies
local function try_extract_code_macro(source, directive_start, code_macros, code_macro_bodies) local function try_extract_code_macro(source, directive_start, code_macros, code_macro_bodies)
local rest = duffle.skip_ws_and_cmt(source, directive_start + 1) local rest = duffle.skip_ws_and_cmt(source, directive_start + 1)
@@ -985,8 +1026,8 @@ end
--- Populates `code_macros` with resolved integer codes AND `code_macro_bodies` with raw RHS text --- Populates `code_macros` with resolved integer codes AND `code_macro_bodies` with raw RHS text
--- (used by the chain walker as cross-source fallback during pass 1b in `M.run`); ignores everything else. --- (used by the chain walker as cross-source fallback during pass 1b in `M.run`); ignores everything else.
--- Used by `M.run` pass 1a to build the cross-source `_code_macros` + `_code_macro_bodies` registries before pass 1b resolves chains. --- Used by `M.run` pass 1a to build the cross-source `_code_macros` + `_code_macro_bodies` registries before pass 1b resolves chains.
--- @param source string --- @param source string
--- @param code_macros table --- @param code_macros table
--- @param code_macro_bodies table --- @param code_macro_bodies table
local function scan_source_pre_pass(source, code_macros, code_macro_bodies) local function scan_source_pre_pass(source, code_macros, code_macro_bodies)
local pos = 1 local pos = 1
@@ -1043,7 +1084,7 @@ local function parse_enum_atom_type_default(body, pos)
if pos > #body then return nil, 0, pos end if pos > #body then return nil, 0, pos end
-- Bare `atom_type` word with word-bounding on both sides. -- Bare `atom_type` word with word-bounding on both sides.
local ident, ident_end = duffle.read_ident(body, pos) local ident, ident_end = duffle.read_ident(body, pos)
if ident ~= "atom_type" then return nil, 0, pos end if ident ~= "atom_type" then return nil, 0, pos end
if pos > 1 then if pos > 1 then
local prev = body:byte(pos - 1) local prev = body:byte(pos - 1)
if duffle.is_alnum_byte(prev) then return nil, 0, pos end if duffle.is_alnum_byte(prev) then return nil, 0, pos end
@@ -1054,19 +1095,19 @@ local function parse_enum_atom_type_default(body, pos)
end end
-- Expect `( ... )` immediately after. -- Expect `( ... )` immediately after.
local open_pos = duffle.skip_ws_and_cmt(body, ident_end) local open_pos = duffle.skip_ws_and_cmt(body, ident_end)
if open_pos > #body or body:sub(open_pos, open_pos) ~= "(" then return nil, 0, pos end if open_pos > #body or body:sub(open_pos, open_pos) ~= "(" then return nil, 0, pos end
local inner, after_close = duffle.read_parens(body, open_pos) local inner, after_close = duffle.read_parens(body, open_pos)
-- Reject any trailing tokens past the close paren other than comma / close-brace (next enum entry / end of enum). -- Reject any trailing tokens past the close paren other than comma / close-brace (next enum entry / end of enum).
local residue = duffle.skip_ws_and_cmt(body, after_close) local residue = duffle.skip_ws_and_cmt(body, after_close)
if residue <= #body then if residue <= #body then
local rbyte = body:byte(residue) local rbyte = body:byte(residue)
if rbyte ~= BYTE_COMMA and rbyte ~= BYTE_CLOSE_BRACE then return nil, 0, pos end if rbyte ~= BYTE_COMMA and rbyte ~= BYTE_CLOSE_BRACE then return nil, 0, pos end
end end
-- Parse the type chain inside the parens (e.g. `V4_S2*` -> ("V4_S2", 1)). -- Parse the type chain inside the parens (e.g. `V4_S2*` -> ("V4_S2", 1)).
local type_name, depth, after_chain = parse_type_chain(inner, 1) local type_name, depth, after_chain = parse_type_chain(inner, 1)
if not type_name then return nil, 0, pos end if not type_name then return nil, 0, pos end
local end_check = duffle.skip_ws_and_cmt(inner, after_chain) local end_check = duffle.skip_ws_and_cmt(inner, after_chain)
if end_check <= #inner then return nil, 0, pos end if end_check <= #inner then return nil, 0, pos end
return type_name, depth, duffle.skip_ws_and_cmt(body, after_close) return type_name, depth, duffle.skip_ws_and_cmt(body, after_close)
end end
@@ -1098,11 +1139,11 @@ end
--- ---
--- Diagnostic-only path: a following `(...)` is recorded as an invalid parenthesized-form marker so the annotation rule can emit a precise "parenthesized form" diagnostic. --- Diagnostic-only path: a following `(...)` is recorded as an invalid parenthesized-form marker so the annotation rule can emit a precise "parenthesized form" diagnostic.
--- The parenthesized form stays diagnostic; the bare form alone carries the runtime stamp. --- The parenthesized form stays diagnostic; the bare form alone carries the runtime stamp.
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer -- source cursor position to resume from --- @return integer -- source cursor position to resume from
local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out) local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
local marker_kind = source:sub(pos, ident_end - 1) local marker_kind = source:sub(pos, ident_end - 1)
@@ -1131,18 +1172,58 @@ local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
return marker_end return marker_end
end end
--- Parse `atom_auto_reg(<atom>, R_<Sym>)` and `phase_auto_reg(<phase>, R_<Sym>)` markers.
---
--- The macros expand to `sym = sym##_Code` per their definition in dsl.atom.h.
--- After preprocessing, the marker renders as a full enum entry of the form `R_<Sym> = R_<Sym>_Code,`.
--- This parser detects the macro invocation site, extracts `(scope_name, sym)`, and stores it
--- in the per-source table (atom_auto_regs or phase_auto_regs) under the scope's name.
---
--- @param source string
--- @param pos integer
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer
local function parse_auto_reg_marker(source, pos, ident_end, line_of, out)
local marker_kind = source:sub(pos, ident_end - 1) -- "atom_auto_reg" or "phase_auto_reg"
local scope_kind = marker_kind == "atom_auto_reg" and "atom" or "phase"
local inner, after_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
local args = duffle.split_top_level_commas(inner)
local scope_name = args[1] and duffle.trim(args[1]) or nil
local sym = args[2] and duffle.trim(args[2]) or nil
-- Filter: only accept `R_<Sym>` form (matches `^R_[%w_]+$`).
if scope_name and sym and sym:match("^R_[%w_]+$") then
if scope_kind == "atom" then
out.atom_auto_regs = out.atom_auto_regs or {}
out.atom_auto_regs[scope_name] = out.atom_auto_regs[scope_name] or {}
out.atom_auto_regs[scope_name][sym] = sym
else
out.phase_auto_regs = out.phase_auto_regs or {}
out.phase_auto_regs[scope_name] = out.phase_auto_regs[scope_name] or {}
out.phase_auto_regs[scope_name][sym] = sym
end
end
return after_paren
end
-- Parse `atom_dbg_reg_default(R_X, <type>...)`; -- Parse `atom_dbg_reg_default(R_X, <type>...)`;
-- the second argument may be a `Type` or `Type*`/`Type**` chain. Records in `out.types[R_X]`. -- the second argument may be a `Type` or `Type*`/`Type**` chain. Records in `out.types[R_X]`.
local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out) local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
local inner, after_paren = read_parens_after(source, ident_end) local inner, after_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end if not inner then return after_paren end
local args = duffle.split_top_level_commas(inner) local args = duffle.split_top_level_commas(inner)
if #args < 1 then if #args < 1 then
-- Annotation pass surfaces this; we still consume the marker. -- Annotation pass surfaces this; we still consume the marker.
return after_paren return after_paren
end end
local reg_name = duffle.trim(args[1]) local reg_name = duffle.trim(args[1])
local type_part = args[2] or "void" local type_part = args[2] or "void"
local type_name, depth = parse_type_chain(type_part, 1) local type_name, depth = parse_type_chain(type_part, 1)
if not type_name then type_name, depth = duffle.trim(type_part), 0 end if not type_name then type_name, depth = duffle.trim(type_part), 0 end
out.types[reg_name] = { out.types[reg_name] = {
@@ -1161,23 +1242,23 @@ local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
end end
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }` --- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer --- @return integer
local function parse_mips_atom(source, pos, ident_end, line_of, out) local function parse_mips_atom(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end) local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end if not inner then return after_paren end
local raw_name = duffle.read_ident(inner, 1) local raw_name = duffle.read_ident(inner, 1)
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start. -- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
local brace_search_pos = after_paren local brace_search_pos = after_paren
local lookahead = duffle.skip_ws_and_cmt(source, after_paren) local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
local look_ident, look_end = duffle.read_ident(source, lookahead) local look_ident, look_end = duffle.read_ident(source, lookahead)
if look_ident == "atom_info" then if look_ident == "atom_info" then
local info_open = duffle.skip_ws_and_cmt(source, look_end) local info_open = duffle.skip_ws_and_cmt(source, look_end)
if source:sub(info_open, info_open) == "(" then if source:sub(info_open, info_open) == "(" then
local info_inner, info_after = duffle.read_parens(source, info_open) local info_inner, info_after = duffle.read_parens(source, info_open)
@@ -1221,7 +1302,7 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
end end
end end
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1) local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
if raw_name and raw_name ~= "" then if raw_name and raw_name ~= "" then
register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source) register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source)
@@ -1231,20 +1312,20 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
end end
--- Parse: `MipsAtomComp_(<name>) { <body> }` --- Parse: `MipsAtomComp_(<name>) { <body> }`
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer --- @return integer
local function parse_mips_atom_comp(source, pos, ident_end, line_of, out) local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end) local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end if not inner then return after_paren end
local raw_name = duffle.read_ident(inner, 1) local raw_name = duffle.read_ident(inner, 1)
if not raw_name then return open_paren + 1 end if not raw_name then return open_paren + 1 end
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
local name = strip_ac_prefix(raw_name) local name = strip_ac_prefix(raw_name)
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source) register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
@@ -1253,14 +1334,14 @@ local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
end end
--- Parse: `MipsAtomComp_Proc_(<name>, { <body> })` — body is inside the LAST `{` in args. --- Parse: `MipsAtomComp_Proc_(<name>, { <body> })` — body is inside the LAST `{` in args.
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer --- @return integer
local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out) local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end) local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end if not inner then return after_paren end
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions). -- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
@@ -1277,21 +1358,63 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
if close_pos > #inner + 1 then return after_paren end if close_pos > #inner + 1 then return after_paren end
local raw_name = inner:match("^%s*([%w_]+)") or "?" local raw_name = inner:match("^%s*([%w_]+)") or "?"
local name = strip_ac_prefix(raw_name) local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{'). -- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source) register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_paren return after_paren
end end
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only). --- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
--- @param source string --- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
--- @param pos integer --- with kind `"atom_proc"` so offsets.lua / components.lua can emit
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
--- @param source string
--- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer
local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
local last_brace_pos = nil
for search_pos = #inner, 1, -1 do
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
end
if not last_brace_pos then return after_paren end
-- Use duffle.read_braces to find the matching close brace.
-- Uses `read_balanced` for delimiter-depth tracking.
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end
-- The atom name is the FIRST ident of the args (matches MipsAtomComp_Proc_'s "first ident" rule).
-- MipsAtom_Proc_ has no `ac_` prefix; `strip_ac_prefix` is a no-op for unprefixed names.
local raw_name = inner:match("^%s*([%w_]+)") or "?"
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
return after_paren
end
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
--- @param source string
--- @param pos integer
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer --- @return integer
local function parse_mips_code(source, pos, ident_end, line_of, out) local function parse_mips_code(source, pos, ident_end, line_of, out)
local next_pos = duffle.skip_ws_and_cmt(source, ident_end) local next_pos = duffle.skip_ws_and_cmt(source, ident_end)
@@ -1300,8 +1423,8 @@ local function parse_mips_code(source, pos, ident_end, line_of, out)
return ident_end return ident_end
end end
local atom_name = next_ident:sub(6) local atom_name = next_ident:sub(6)
local body, after_brace, body_off = find_body_braces(source, next_after, ident_end) local body, after_brace, body_off = find_body_braces(source, next_after, ident_end)
if not body then return after_brace end if not body then return after_brace end
register_raw_atom(out, line_of(pos), atom_name, body, body_off, atom_name, pos) register_raw_atom(out, line_of(pos), atom_name, body, body_off, atom_name, pos)
@@ -1320,13 +1443,13 @@ end
--- pointer_depth = 0 --- pointer_depth = 0
--- } -- byte_size + per-field offset/byte_size set by the propagation pass. --- } -- byte_size + per-field offset/byte_size set by the propagation pass.
--- Also populates `out.binds[]` IFF `name:sub(1, 6) == "Binds_"`. --- Also populates `out.binds[]` IFF `name:sub(1, 6) == "Binds_"`.
--- @param body string --- @param body string
--- @param name string --- @param name string
--- @param pos integer --- @param pos integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
local function register_struct_type(body, name, pos, line_of, out) local function register_struct_type(body, name, pos, line_of, out)
local fields = parse_struct_body_fields(body) local fields = parse_struct_body_fields(body)
local source_pos = line_of(pos) local source_pos = line_of(pos)
out.type_name_registry[name] = { out.type_name_registry[name] = {
name = name, name = name,
@@ -1352,11 +1475,11 @@ end
--- Register an Enum_ entry in type_name_registry. --- Register an Enum_ entry in type_name_registry.
--- Local helper for parse_typedef_binds. Captures the underlying type (1st arg of `Enum_(<underlying>, <name>)`) and the body fields. --- Local helper for parse_typedef_binds. Captures the underlying type (1st arg of `Enum_(<underlying>, <name>)`) and the body fields.
--- @param underlying string --- @param underlying string
--- @param name string --- @param name string
--- @param body string --- @param body string
--- @param pos integer --- @param pos integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
local function register_enum_type(underlying, name, body, pos, line_of, out) local function register_enum_type(underlying, name, body, pos, line_of, out)
local fields = parse_enum_body_fields(body) local fields = parse_enum_body_fields(body)
out.type_name_registry[name] = { out.type_name_registry[name] = {
@@ -1375,18 +1498,18 @@ end
--- Captures the underlying type ident (LHS of `typedef <type> <alias>;`) and exposes it through the registry. --- Captures the underlying type ident (LHS of `typedef <type> <alias>;`) and exposes it through the registry.
--- The propagation pass follows the underlying_type chain to resolve byte_size. --- The propagation pass follows the underlying_type chain to resolve byte_size.
--- @param underlying string --- @param underlying string
--- @param name string --- @param name string
--- @param pos integer --- @param pos integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
local function register_typedef_alias(underlying, name, pos, line_of, out) local function register_typedef_alias(underlying, name, pos, line_of, out)
out.type_name_registry[name] = { out.type_name_registry[name] = {
name = name, name = name,
kind = "typedef", kind = "typedef",
underlying_type = underlying, underlying_type = underlying,
source_line = line_of(pos), source_line = line_of(pos),
source_file = out._source_file, source_file = out._source_file,
pointer_depth = 0, pointer_depth = 0,
} }
end end
@@ -1396,29 +1519,29 @@ end
--- 1. `typedef Struct_(<name>) { <body> } <alias>;` adds to type_name_registry (kind="struct"). --- 1. `typedef Struct_(<name>) { <body> } <alias>;` adds to type_name_registry (kind="struct").
--- Binds_* aliases also land in out.binds[]. --- Binds_* aliases also land in out.binds[].
--- 2. `typedef Enum_(<underlying>, <name>) { <body> } <alias>;` --- 2. `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
--- adds to type_name_registry (kind="enum"). --- Adds to type_name_registry (kind="enum").
--- 3. `typedef <type> <alias>;` simple typedef alias. --- 3. `typedef <type> <alias>;` simple typedef alias.
--- Adds to type_name_registry (kind="typedef"). --- Adds to type_name_registry (kind="typedef").
--- 4. `typedef <type> TSet_(<name>);` duffle TSet_ convention. --- 4. `typedef <type> TSet_(<name>);` duffle TSet_ convention.
--- Strips TSet_ wrapper; adds to type_name_registry (kind="typedef") with underlying_type=<type>. --- Strips TSet_ wrapper; adds to type_name_registry (kind="typedef") with underlying_type=<type>.
--- ---
--- All four shapes also attach an "unrelated" debug-skip marker (the existing behavior — typedef declarations don't carry atom_dbg_skip). --- All four shapes also attach an "unrelated" debug-skip marker (the existing behavior — typedef declarations don't carry atom_dbg_skip).
--- @param source string --- @param source string
--- @param pos integer --- @param pos integer
--- @param ident_end integer --- @param ident_end integer
--- @param line_of fun(pos: integer): integer --- @param line_of fun(pos: integer): integer
--- @param out SourceScan --- @param out SourceScan
--- @return integer --- @return integer
local function parse_typedef_binds(source, pos, ident_end, line_of, out) local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local after_typedef = duffle.skip_ws_and_cmt(source, ident_end) local after_typedef = duffle.skip_ws_and_cmt(source, ident_end)
local id2, id2_end = duffle.read_ident(source, after_typedef) local id2, id2_end = duffle.read_ident(source, after_typedef)
if not id2 then return ident_end end if not id2 then return ident_end end
-- ── Shape 1: `typedef Struct_(<name>) { <body> } <alias>;` ──────────── -- ── Shape 1: `typedef Struct_(<name>) { <body> } <alias>;` ────────────
if id2 == "Struct_" then if id2 == "Struct_" then
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end) local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end)
if not inner then return id2_end end if not inner then return id2_end end
local name = duffle.trim(inner) local name = duffle.trim(inner)
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
@@ -1428,7 +1551,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
-- ── Shape 2: `typedef Enum_(<underlying>, <name>) { <body> } <alias>;` -- ── Shape 2: `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
elseif id2 == "Enum_" then elseif id2 == "Enum_" then
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end) local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end)
if not inner then return id2_end end if not inner then return id2_end end
-- Split `inner` on the first top-level comma into (<underlying>, <name>). -- Split `inner` on the first top-level comma into (<underlying>, <name>).
local args = duffle.split_top_level_commas(inner) local args = duffle.split_top_level_commas(inner)
@@ -1436,7 +1559,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local underlying = duffle.trim(args[1]) local underlying = duffle.trim(args[1])
local name = duffle.trim(args[2]) local name = duffle.trim(args[2])
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1) local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end if not body then return after_brace end
register_enum_type(underlying, name, body, pos, line_of, out) register_enum_type(underlying, name, body, pos, line_of, out)
attach_debug_skip_marker(out, "unrelated") attach_debug_skip_marker(out, "unrelated")
@@ -1469,11 +1592,10 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
-- Shape 4 (TSet_ at id2 position): no preceding underlying span. -- Shape 4 (TSet_ at id2 position): no preceding underlying span.
if id2 == "TSet_" then if id2 == "TSet_" then
local inner, after_paren = read_parens_after(source, id2_end, id2_end) local inner, after_paren = read_parens_after(source, id2_end, id2_end)
if not inner then return id2_end end if not inner then return id2_end end
local tset_name = duffle.trim(inner) local tset_name = duffle.trim(inner)
-- Empty underlying span is acceptable; the TSet_ wrapper itself -- Empty underlying span is acceptable; the TSet_ wrapper itself encodes the alias identity (per the duffle TSet_ convention).
-- encodes the alias identity (per the duffle TSet_ convention).
register_typedef_alias("", tset_name, pos, line_of, out) register_typedef_alias("", tset_name, pos, line_of, out)
attach_debug_skip_marker(out, "unrelated") attach_debug_skip_marker(out, "unrelated")
return after_paren return after_paren
@@ -1491,13 +1613,13 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
while scan < semi_pos do while scan < semi_pos do
scan = duffle.skip_ws_and_cmt(source, scan) scan = duffle.skip_ws_and_cmt(source, scan)
if scan >= semi_pos then break end if scan >= semi_pos then break end
local id, id_end = duffle.read_ident(source, scan) local id, id_end = duffle.read_ident(source, scan)
if not id then if not id then
scan = scan + 1 scan = scan + 1
elseif id == "TSet_" then elseif id == "TSet_" then
-- Shape 4 (TSet_ at non-id2 position): grab the parenthesized argument. -- Shape 4 (TSet_ at non-id2 position): grab the parenthesized argument.
local inner, after_paren = read_parens_after(source, id_end, id_end) local inner, after_paren = read_parens_after(source, id_end, id_end)
if inner then if inner then
tset_arg = duffle.trim(inner) tset_arg = duffle.trim(inner)
tset_arg_end = after_paren tset_arg_end = after_paren
tset_pos = scan tset_pos = scan
@@ -1607,6 +1729,16 @@ local function parse_enum_entry(source, body, body_offset, line_of, out, entry_n
local value, value_end = parse_enum_value(body, after_ws, out) local value, value_end = parse_enum_value(body, after_ws, out)
if value == nil then return value_start end if value == nil then return value_start end
-- Capture the trailing C-comment (if any) before `skip_ws_and_cmt` discards it.
-- The `atom_auto_reg(<scope>, <sym>)` macro expands to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
-- so the scope name lives in the comment after the RHS value. Routes through `out.atom_entry_comments`
-- for downstream `parse_enum` to split into `out.atom_auto_regs` / `out.phase_auto_regs`.
local trailing_cmt = read_trailing_cmt_after(body, value_end)
if trailing_cmt then
out.atom_entry_comments = out.atom_entry_comments or {}
out.atom_entry_comments[entry_name] = trailing_cmt
end
local after_value = duffle.skip_ws_and_cmt(body, value_end) local after_value = duffle.skip_ws_and_cmt(body, value_end)
local has_atom_reg, end_after_atom_reg = check_bare_atom_reg(body, after_value) local has_atom_reg, end_after_atom_reg = check_bare_atom_reg(body, after_value)
@@ -1662,15 +1794,24 @@ local function parse_enum_body(source, body, body_offset, line_of, out)
else else
local entry_name, name_end = duffle.read_ident(body, pos) local entry_name, name_end = duffle.read_ident(body, pos)
if entry_name then if entry_name then
local after_name = duffle.skip_ws_and_cmt(body, name_end) -- In-enum `atom_auto_reg(<scope>, R_<Sym>)` / `phase_auto_reg(<scope>, R_<Sym>)` markers:
if body:byte(after_name) == BYTE_EQUAL then -- the C preprocessor expands them to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
local new_pos = parse_enum_entry( -- but the metaprogram reads source-as-written so we must dispatch the parser here too.
source, body, body_offset, line_of, out, -- Mirrors the top-level `DECL_PARSERS` entry for `atom_auto_reg` / `phase_auto_reg`.
entry_name, pos, after_name + 1 if entry_name == "atom_auto_reg" or entry_name == "phase_auto_reg" then
) local new_pos = parse_auto_reg_marker(body, pos, name_end, line_of, out)
if new_pos > pos then pos = new_pos else pos = after_name + 1 end if new_pos > pos then pos = new_pos else pos = name_end end
else else
pos = name_end local after_name = duffle.skip_ws_and_cmt(body, name_end)
if body:byte(after_name) == BYTE_EQUAL then
local new_pos = parse_enum_entry(
source, body, body_offset, line_of, out,
entry_name, pos, after_name + 1
)
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
else
pos = name_end
end
end end
else else
pos = pos + 1 pos = pos + 1
@@ -1700,6 +1841,25 @@ local function parse_enum(source, pos, ident_end, line_of, out)
if not body then return after_brace end if not body then return after_brace end
parse_enum_body(source, body, body_off, line_of, out) parse_enum_body(source, body, body_off, line_of, out)
-- Route `atom_auto_reg:` / `phase_auto_reg:` markers discovered in trailing C-comments
-- into the per-source `atom_auto_regs` / `phase_auto_regs` projections.
-- Pattern matches the RHS expansion `R_<Sym> = R_<Sym>_Code /* <kind>_auto_reg: <scope> */`
-- emitted by the `atom_auto_reg` / `phase_auto_reg` macros in dsl.atom.h.
for entry_name, cmt_text in pairs(out.atom_entry_comments or {}) do
local atom_scope = cmt_text:match("atom_auto_reg:%s*([%w_]+)")
if atom_scope then
out.atom_auto_regs = out.atom_auto_regs or {}
out.atom_auto_regs[atom_scope] = out.atom_auto_regs[atom_scope] or {}
out.atom_auto_regs[atom_scope][entry_name] = entry_name
end
local phase_scope = cmt_text:match("phase_auto_reg:%s*([%w_]+)")
if phase_scope then
out.phase_auto_regs = out.phase_auto_regs or {}
out.phase_auto_regs[phase_scope] = out.phase_auto_regs[phase_scope] or {}
out.phase_auto_regs[phase_scope][entry_name] = entry_name
end
end
return after_brace return after_brace
end end
@@ -1713,12 +1873,18 @@ end
local DECL_PARSERS = { local DECL_PARSERS = {
MipsAtom_ = parse_mips_atom, MipsAtom_ = parse_mips_atom,
MipsAtom_Proc_ = parse_mips_atom_proc,
MipsAtomComp_ = parse_mips_atom_comp, MipsAtomComp_ = parse_mips_atom_comp,
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc, MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other -- `atom_dbg_skip` is the only debug-skip parser entry. Every other
-- identifier follows the ordinary unrelated-token path; there is no alias. -- identifier follows the ordinary unrelated-token path; there is no alias.
atom_dbg_skip = parse_dbg_skip_marker, atom_dbg_skip = parse_dbg_skip_marker,
atom_dbg_reg_default = parse_atom_dbg_reg_default, atom_dbg_reg_default = parse_atom_dbg_reg_default,
-- `atom_auto_reg(atom, R_<Sym>)` and `phase_auto_reg(phase, R_<Sym>)` populate per-source
-- `out.atom_auto_regs` / `out.phase_auto_regs`; the cross-source merge lands in
-- `corpus.atom_auto_regs` / `corpus.phase_auto_regs` (first-wins).
atom_auto_reg = parse_auto_reg_marker,
phase_auto_reg = parse_auto_reg_marker,
MipsCode = parse_mips_code, MipsCode = parse_mips_code,
typedef = parse_typedef_binds, typedef = parse_typedef_binds,
_Pragma = parse_pragma_macro, _Pragma = parse_pragma_macro,
@@ -1753,6 +1919,14 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
debug_skip_markers = {}, debug_skip_markers = {},
types = {}, types = {},
atom_views = {}, atom_views = {},
-- Per-source projection for `atom_auto_reg(<atom>, R_<Sym>)` markers.
-- Each entry is keyed by atom_name; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
-- Merged cross-source into `corpus.atom_auto_regs` (first-wins).
atom_auto_regs = {},
-- Per-source projection for `phase_auto_reg(<phase>, R_<Sym>)` markers.
-- Each entry is keyed by phase_label; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
-- Merged cross-source into `corpus.phase_auto_regs` (first-wins).
phase_auto_regs = {},
line_of = line_of, line_of = line_of,
-- Source-derived register-alias registry (atom_reg opt-in entries). -- Source-derived register-alias registry (atom_reg opt-in entries).
-- Keys are full R_* idents (never stripped); see parse_enum / parse_enum_body. -- Keys are full R_* idents (never stripped); see parse_enum / parse_enum_body.
@@ -1957,7 +2131,7 @@ local function merge_named_with_sites(registry, name, new_entry, site, collision
registry[name].sites = { site } registry[name].sites = { site }
return return
end end
local existing = registry[name] local existing = registry[name]
local new_shape = shape_fn(new_entry) local new_shape = shape_fn(new_entry)
local old_shape = shape_fn(existing) local old_shape = shape_fn(existing)
if new_shape == old_shape and new_shape ~= "" then if new_shape == old_shape and new_shape ~= "" then
@@ -1992,6 +2166,8 @@ local function merge_corpus_registries(corpus)
corpus.atom_ctxs = corpus.atom_ctxs or {} corpus.atom_ctxs = corpus.atom_ctxs or {}
corpus.atom_phases = corpus.atom_phases or {} corpus.atom_phases = corpus.atom_phases or {}
corpus.atom_infos = corpus.atom_infos or {} corpus.atom_infos = corpus.atom_infos or {}
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
corpus.collisions = corpus.collisions or {} corpus.collisions = corpus.collisions or {}
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge). -- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
@@ -2035,7 +2211,7 @@ local function merge_corpus_registries(corpus)
corpus.collisions, "binds", bind_shape) corpus.collisions, "binds", bind_shape)
end end
-- atoms_by_name: MipsAtom_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name). -- atoms_by_name: MipsAtom_(name) + MipsAtom_Proc_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
-- Each atom carries `{line, name, body, body_off, kind, raw_name, ...}`. -- Each atom carries `{line, name, body, body_off, kind, raw_name, ...}`.
-- Duplicate atom names across sources are first-wins + collision; see the atom_infos block below for the evidence list. -- Duplicate atom names across sources are first-wins + collision; see the atom_infos block below for the evidence list.
for _, atom_entry in ipairs(scan.atoms or {}) do for _, atom_entry in ipairs(scan.atoms or {}) do
@@ -2070,6 +2246,22 @@ local function merge_corpus_registries(corpus)
corpus.collisions, "phase", phase_shape) corpus.collisions, "phase", phase_shape)
end end
-- atom_auto_regs: keyed by atom scope name; each carries a `{R_<Sym> = R_<Sym>}` map.
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
for atom_scope, syms in pairs(scan.atom_auto_regs or {}) do
if corpus.atom_auto_regs[atom_scope] == nil then
corpus.atom_auto_regs[atom_scope] = syms
end
end
-- phase_auto_regs: keyed by phase label; each carries a `{R_<Sym> = R_<Sym>}` map.
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
for phase_label, syms in pairs(scan.phase_auto_regs or {}) do
if corpus.phase_auto_regs[phase_label] == nil then
corpus.phase_auto_regs[phase_label] = syms
end
end
-- atom_infos: ALWAYS append every record in source/declaration order. -- atom_infos: ALWAYS append every record in source/declaration order.
-- Duplicates are preserved so the annotation pass can flag them via `check_unique_annotation`; -- Duplicates are preserved so the annotation pass can flag them via `check_unique_annotation`;
-- The merge is purely order-preserving. -- The merge is purely order-preserving.
File diff suppressed because it is too large Load Diff
+1 -1
View File
@@ -93,7 +93,7 @@ end
--- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`. --- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
--- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs. --- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
--- Current component counts are computed and inserted by `passes/components.lua` --- Current component counts are computed and inserted by `passes/components.lua`
--- after the components pass iterates `corpus.source_order` and writes each source's `<dir_basename>.macs.h` file. --- after the components pass iterates `corpus.source_order` and writes each source-directory's `gen/macs.h` file.
--- ---
--- Contract: --- Contract:
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership). --- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
Binary file not shown.
+52 -59
View File
@@ -2,24 +2,21 @@
--- ---
--- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection). --- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection).
--- ---
--- **Architecture**: --- Architecture:
--- - **PASSES table** — declarative dep graph (data, not code). --- - PASSES table: Declarative dep graph (data, not code).
--- - **FLAG_HANDLERS table** — maps CLI flags to handlers. --- - FLAG_HANDLERS table: Maps CLI flags to handlers.
--- - **parse_args****build_ctx** (resolves unity/direct includes or exact sources; no semantic scanning) → **topo_sort****dispatch_passes**. --- - parse_args → build_ctx (resolves unity/direct includes or exact sources) → topo_sort → dispatch_passes.
--- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`). --- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`).
--- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`. --- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`.
--- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only. --- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only.
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
---
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Bootstrap: load `duffle_paths.lua` via this script's own path. -- Bootstrap: load `duffle_paths.lua` via this script's own path.
-- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua"); -- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua");
-- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path, not ours). -- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path).
-- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`. -- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`.
-- So the dofile's return value is the duffle module. -- So the dofile's return value is the duffle module.
local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil
@@ -57,45 +54,44 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__"
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
--- @class PassDescriptor --- @class PassDescriptor
--- @field module string -- module name passed to require() --- @field module string -- Module name passed to require()
--- @field kind string -- "shared" | "header-output" | "validation" | "diagnostic" | "report" --- @field kind string -- "shared" | "header-output" | "validation" | "diagnostic" | "report"
--- -- Report severity is independent from process exit policy (see PASS_KIND_STOP_ON_ERROR). --- -- Report severity is independent from process exit policy (see PASS_KIND_STOP_ON_ERROR).
--- @field deps string[] -- names of upstream passes --- @field deps string[] -- Names of upstream passes
--- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of --- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
--- -- (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
--- @class SourceFile --- @class SourceFile
--- @field path string -- absolute path to the source file --- @field path string -- Absolute path to the source file
--- @field text string -- the full source text --- @field text string -- Full source text
--- @field dir string -- the directory containing the source --- @field dir string -- Directory containing the source
--- @field basename string -- filename without extension --- @field basename string -- Filename without extension
--- @class PassCtx --- @class PassCtx
--- @field metadata_path string -- path to word_count.metadata.h --- @field metadata_path string -- Path to word_count.metadata.h
--- @field shared table -- cross-pass shared state --- @field shared table -- Cross-pass shared state
--- @field shared.corpus table -- canonical authored-source/project projection --- @field shared.corpus table -- Authored-source/project projection
--- @field out_root string -- output root (e.g. "build/gen") --- @field out_root string -- Output root (e.g. "build/gen")
--- @field project_root string -- PS1 repository root --- @field project_root string -- PS1 repository root
--- @field flags table -- CLI flags + per-pass stash --- @field flags table -- CLI flags + per-pass stash
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- If true, log diagnostic info
--- @class Finding --- @class Finding
--- @field line integer -- source line (or 0 for pass-level) --- @field line integer -- Source line (or 0 for pass-level)
--- @field msg string -- finding message --- @field msg string -- Finding message
--- @class PassResult --- @class PassResult
--- @field outputs PassOutputEntry[] -- emitted file paths --- @field outputs PassOutputEntry[] -- Emitted file paths
--- @field errors Finding[] -- build-stops (per-pass kind policy) --- @field errors Finding[] -- Build-stops (per-pass kind policy)
--- @field warnings Finding[] -- informational --- @field warnings Finding[] -- Informational
--- @class ParsedArgs --- @class ParsedArgs
--- @field requested_set string[] -- pass names to run (explicit --all expanded) --- @field requested_set string[] -- Pass names to run (explicit --all expanded)
--- @field sources string[] -- exact --source values, retained in CLI order --- @field sources string[] -- Exact --source values, retained in CLI order
--- @field unity_root string|nil -- --unity-root value; mutually exclusive with sources --- @field unity_root string|nil -- --unity-root value; mutually exclusive with sources
--- @field metadata string -- --metadata value --- @field metadata string -- --metadata value
--- @field out_root string -- --out-root value (default "build/gen") --- @field out_root string -- --out-root value (default "build/gen")
--- @field project_root string -- PS1 repository root (derived from metadata by default) --- @field project_root string -- PS1 repository root (derived from metadata by default)
--- @field verbose boolean -- if true, log diagnostic info --- @field verbose boolean -- If true, log diagnostic info
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- PASSES Table -- PASSES Table
@@ -122,6 +118,12 @@ local PASSES = {
kind = "header-output", kind = "header-output",
deps = {"scan-source", "word-counts"}, deps = {"scan-source", "word-counts"},
}, },
auto_reg = {
module = "passes.auto_reg",
kind = "header-output",
deps = {"components"},
groups = { "pre-link" },
},
["emission-model"] = { ["emission-model"] = {
module = "passes.emission_model", module = "passes.emission_model",
kind = "validation", kind = "validation",
@@ -141,7 +143,7 @@ local PASSES = {
["static-analysis"] = { ["static-analysis"] = {
module = "passes.static_analysis", module = "passes.static_analysis",
-- "diagnostic" — every `error`/`warning` finding is written to the report file; -- "diagnostic" — every `error`/`warning` finding is written to the report file;
-- the orchestrator does NOT exit non-zero on these findings (see PASS_KIND_STOP_ON_ERROR). -- The orchestrator does NOT exit non-zero on these findings (see PASS_KIND_STOP_ON_ERROR).
-- Report severity is independent from process exit policy. -- Report severity is independent from process exit policy.
kind = "diagnostic", kind = "diagnostic",
deps = {"scan-source", "word-counts", "components", "emission-model"}, deps = {"scan-source", "word-counts", "components", "emission-model"},
@@ -160,18 +162,18 @@ local PASSES = {
report = { report = {
module = "passes.report", module = "passes.report",
kind = "report", kind = "report",
deps = {"annotation", "static-analysis"}, deps = {"annotation", "static-analysis", "atoms-source-map"}, -- +atoms-source-map (consolidated-report-files refactor, 2026-07-26)
groups = { "pre-link" }, groups = { "pre-link" },
}, },
} }
-- ──────────────────────────────────────────────────────────────────────────── -- ────────────────────────────────────────────────────────────────────────────
-- Phase-root selection: derive the sorted set of roots belonging to a named build-phase group, then append them to `args.requested_set`. -- Phase-root selection: Derive the sorted set of roots belonging to a named build-phase group, then append them to `args.requested_set`.
-- topo_sort closes the transitive deps from there; dispatch_passes runs every resolved pass without phase-filtering. -- topo_sort closes the transitive deps from there; dispatch_passes runs every resolved pass without phase-filtering.
-- ──────────────────────────────────────────────────────────────────────────── -- ────────────────────────────────────────────────────────────────────────────
--- @param group_name string -- the build-phase group ("pre-link" | "post-link") --- @param group_name string -- Build-phase group ("pre-link" | "post-link")
--- @return string[] -- sorted root pass names belonging to that group --- @return string[] -- Sorted root pass names belonging to that group
local function roots_for_group(group_name) local function roots_for_group(group_name)
local names = {} local names = {}
for name, pass in pairs(PASSES) do for name, pass in pairs(PASSES) do
@@ -206,9 +208,10 @@ end
-- Pass-kind taxonomy: Which kinds stop the build on errors? -- Pass-kind taxonomy: Which kinds stop the build on errors?
-- --
-- Report severity is independent from process exit policy. A "diagnostic" pass still writes every `error`/`warning` finding into its report file, -- Report severity is independent from process exit policy.
-- A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
-- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero. -- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero.
-- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true". -- Adding a new pass kind requires listing it here explicitly; An unknown kind must not silently fall back to "true".
local PASS_KIND_STOP_ON_ERROR = { local PASS_KIND_STOP_ON_ERROR = {
["shared"] = false, ["shared"] = false,
["header-output"] = true, ["header-output"] = true,
@@ -218,8 +221,7 @@ local PASS_KIND_STOP_ON_ERROR = {
} }
-- Closed set of CLI flags -> pass names. -- Closed set of CLI flags -> pass names.
-- Per-pass flags (e.g. --word-counts) live here; phase flags (--pre-link, --post-link, --all) -- Per-pass flags (e.g. --word-counts); phase flags (--pre-link, --post-link, --all) are within FLAG_HANDLERS because they own side effects or invoke group-derivation logic.
-- live in FLAG_HANDLERS because they own side effects or invoke group-derivation logic.
-- dwarf-injection is *also* a per-pass opt-in flag, but its selection + opt-in state are both owned by the explicit FLAG_HANDLERS entry below -- dwarf-injection is *also* a per-pass opt-in flag, but its selection + opt-in state are both owned by the explicit FLAG_HANDLERS entry below
-- (it sets args.flags.dwarf_injection and appends "dwarf-injection" to requested_set), so it is intentionally absent from this table. -- (it sets args.flags.dwarf_injection and appends "dwarf-injection" to requested_set), so it is intentionally absent from this table.
local PASS_FLAG_TO_NAME = { local PASS_FLAG_TO_NAME = {
@@ -248,7 +250,6 @@ end
-- Per-flag handlers. Each handler takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it). -- Per-flag handlers. Each handler takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
-- Returning nil + os.exit() handles termination flags (--help). -- Returning nil + os.exit() handles termination flags (--help).
local FLAG_HANDLERS = {} local FLAG_HANDLERS = {}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -274,9 +275,9 @@ PASS_FLAGS:
Or pick any subset: Or pick any subset:
--scan-source Scan sources into the fat SourceScan payload --scan-source Scan sources into the fat SourceScan payload
--word-counts Load metadata.h + scan for existing .macs.h --word-counts Load metadata.h + scan for existing .macs.h
--components Generate <module>/gen/<basename>.macs.h --components Generate <srcdir>/gen/macs.h (per-directory aggregation)
--validate Run atom annotation DSL validation --validate Run atom annotation DSL validation
--offsets Generate <module>/gen/<basename>.offsets.h --offsets Generate <srcdir>/gen/offsets.h (per-directory aggregation)
--atoms-source-map Generate <basename>.atoms.sourcemap.txt per source --atoms-source-map Generate <basename>.atoms.sourcemap.txt per source
--dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf. --dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf.
--static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget --static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget
@@ -319,8 +320,7 @@ local function require_flag_value(argv, arg_idx, flag)
local next_known = type(value) == "string" local next_known = type(value) == "string"
and (FLAG_HANDLERS[value] ~= nil or PASS_FLAG_TO_NAME[value] ~= nil) and (FLAG_HANDLERS[value] ~= nil or PASS_FLAG_TO_NAME[value] ~= nil)
if value == nil or next_known then if value == nil or next_known then
io.stderr:write("ps1_meta: " .. flag .. " requires " io.stderr:write("ps1_meta: " .. flag .. " requires " .. FLAG_VALUE_NAMES[flag] .. "\n")
.. FLAG_VALUE_NAMES[flag] .. "\n")
os.exit(EXIT_INTERNAL_ERROR) os.exit(EXIT_INTERNAL_ERROR)
end end
return value, arg_idx + 1 return value, arg_idx + 1
@@ -330,12 +330,10 @@ end
-- Termination flags like --help call os.exit() instead. -- Termination flags like --help call os.exit() instead.
-- Populated AFTER print_help so the --help handler can reference it as an upvalue (Lua resolves locals at closure-call time, -- Populated AFTER print_help so the --help handler can reference it as an upvalue (Lua resolves locals at closure-call time,
-- but if the closure is defined before the local, it falls back to _G). -- but if the closure is defined before the local, it falls back to _G).
FLAG_HANDLERS["--help"] = function(args)
print_help()
os.exit(0)
end
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end FLAG_HANDLERS["--help"] = function(args) print_help(); os.exit(0) end
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end
FLAG_HANDLERS["--source"] = function(args, argv, arg_idx) FLAG_HANDLERS["--source"] = function(args, argv, arg_idx)
local value, value_idx = require_flag_value(argv, arg_idx, "--source") local value, value_idx = require_flag_value(argv, arg_idx, "--source")
args.sources[#args.sources + 1] = value args.sources[#args.sources + 1] = value
@@ -492,14 +490,13 @@ end
--- @param args ParsedArgs --- @param args ParsedArgs
--- @return PassCtx --- @return PassCtx
local function build_ctx(args) local function build_ctx(args)
local normalized_project_root = duffle.normalize_path(args.project_root) local normalized_project_root = duffle.normalize_path(args.project_root)
local project_root = normalized_project_root local project_root = normalized_project_root
local project_root_is_absolute = normalized_project_root:match("^%a:/") local project_root_is_absolute = normalized_project_root:match("^%a:/")
or normalized_project_root:sub(1, 2) == "//" or normalized_project_root:sub(1, 2) == "//"
or normalized_project_root:sub(1, 1) == "/" or normalized_project_root:sub(1, 1) == "/"
if not project_root_is_absolute then if not project_root_is_absolute then
-- canonical_path_key validates ordinary relative paths and rejects -- canonical_path_key validates ordinary relative paths and rejects drive-relative paths before the absolute-path rewrite is performed.
-- drive-relative paths before the absolute-path rewrite is performed.
duffle.canonical_path_key(normalized_project_root) duffle.canonical_path_key(normalized_project_root)
project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root)) project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root))
else else
@@ -513,8 +510,7 @@ local function build_ctx(args)
project_root = project_root, project_root = project_root,
}) })
if not ok_resolve then if not ok_resolve then
io.stderr:write("ps1_meta: cannot resolve --unity-root " io.stderr:write("ps1_meta: cannot resolve --unity-root " .. tostring(args.unity_root) .. ": " .. tostring(resolved) .. "\n")
.. tostring(args.unity_root) .. ": " .. tostring(resolved) .. "\n")
os.exit(EXIT_INTERNAL_ERROR) os.exit(EXIT_INTERNAL_ERROR)
end end
resolution = resolved resolution = resolved
@@ -530,8 +526,7 @@ local function build_ctx(args)
local path = duffle.normalize_path(input_path) local path = duffle.normalize_path(input_path)
local key_ok, key_or_error = pcall(duffle.canonical_path_key, path) local key_ok, key_or_error = pcall(duffle.canonical_path_key, path)
if not key_ok then if not key_ok then
error("ps1_meta: invalid --source " .. input_path .. ": " error("ps1_meta: invalid --source " .. input_path .. ": " .. tostring(key_or_error), 0)
.. tostring(key_or_error), 0)
end end
local file = io.open(path, "r") local file = io.open(path, "r")
if not file then if not file then
@@ -629,9 +624,7 @@ local function topo_sort(passes, requested_set)
changed = false changed = false
for name, _ in pairs(needed) do for name, _ in pairs(needed) do
local pass = passes[name] local pass = passes[name]
if not pass then if not pass then error("unknown pass '" .. name .. "' requested") end
error("unknown pass '" .. name .. "' requested")
end
for _, dep in ipairs(pass.deps) do for _, dep in ipairs(pass.deps) do
if not needed[dep] then if not needed[dep] then
needed[dep] = true needed[dep] = true
+11 -6
View File
@@ -14,16 +14,21 @@ $url_armips = 'https://github.com/Kingcom/armips.git'
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git' $url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git' $url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git' $url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
$path_armips = join-path $path_toolchain 'armips' $path_armips = join-path $path_toolchain 'armips'
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux' $path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu' $path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
$path_lpeg = join-path $path_toolchain 'lpeg' $path_lpeg = join-path $path_toolchain 'lpeg'
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
clone-gitrepo $path_armips $url_armips clone-gitrepo $path_armips $url_armips
clone-gitrepo $path_lpeg $url_lpeg clone-gitrepo $path_lpeg $url_lpeg
clone-gitrepo $path_pcsx_redux $url_pcsx_redux clone-gitrepo $path_pcsx_redux $url_pcsx_redux
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
$path_armips_build = join-path $path_armips 'build' $path_armips_build = join-path $path_armips 'build'
verify-path $path_armips_build verify-path $path_armips_build
@@ -56,7 +61,7 @@ if (-not $msbuild_exe) {
} }
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln' $path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal & $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim; # Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
# we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers). # we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers).
@@ -70,8 +75,8 @@ if (-not $luajit_prefix -or -not (Test-Path (Join-Path $luajit_prefix 'bin/luaji
# Discover the luajit include dir by globbing `include/luajit-*`. # Discover the luajit include dir by globbing `include/luajit-*`.
# This avoids hardcoding a specific version (e.g. `luajit-2.1`). # This avoids hardcoding a specific version (e.g. `luajit-2.1`).
$luajit_include_root = Join-Path $luajit_prefix 'include' $luajit_include_root = Join-Path $luajit_prefix 'include'
$lua_inc_dir = Get-ChildItem -Path $luajit_include_root -Directory -Filter 'luajit-*' -ErrorAction SilentlyContinue | $lua_inc_dir = Get-ChildItem -Path $luajit_include_root -Directory -Filter 'luajit-*' -ErrorAction SilentlyContinue |
Select-Object -First 1 -ExpandProperty FullName Select-Object -First 1 -ExpandProperty FullName
if (-not $lua_inc_dir) { if (-not $lua_inc_dir) {
write-error "No 'luajit-*' include dir found under '$luajit_include_root'. The scoop luajit install may be broken." write-error "No 'luajit-*' include dir found under '$luajit_include_root'. The scoop luajit install may be broken."
exit 1 exit 1
@@ -90,7 +95,7 @@ $lpeg_compile_args = @(
'-o', 'lpeg.dll' '-o', 'lpeg.dll'
) + $lpeg_sources + @('-lluajit-5.1') ) + $lpeg_sources + @('-lluajit-5.1')
push-location $path_lpeg push-location $path_lpeg
& gcc @lpeg_compile_args & gcc @lpeg_compile_args
pop-location pop-location
# ════════════════════════════════════════════════════════════════════════════ # ════════════════════════════════════════════════════════════════════════════
@@ -101,8 +106,8 @@ pop-location
$path_lfs = join-path $path_toolchain 'lfs' $path_lfs = join-path $path_toolchain 'lfs'
verify-path $path_lfs verify-path $path_lfs
$lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c' $lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c'
$lfs_dll = join-path $path_lfs 'lfs.dll' $lfs_dll = join-path $path_lfs 'lfs.dll'
$lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a' $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
& gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import & gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import