Author SHA1 Message Date
ed 54a5bb9a31 starting to optimize 2026-08-04 12:59:51 -04:00
ed e0f4ac873d spamming load delay slots for now as a fix... 2026-08-04 09:07:53 -04:00
ed f17fa9165e wip: input was working... messed it up (bios snapshot reads) 2026-08-04 00:50:12 -04:00
ed 8282f8e902 overkill sio cruft, not keeping. 2026-08-03 10:12:06 -04:00
ed 9eb696ece8 drafting 2026-08-02 21:58:57 -04:00
ed 858e57f293 preparing to overhaul input handling 2026-08-02 17:49:24 -04:00
ed afcd9b86f0 Gaining clarity on tape abi.. screen_init atoms done. Time to finish rest of joypad course vods... 2026-08-02 15:19:52 -04:00
ed 43cd4e0344 WIP: working towards minimizing C-ABI & PsyQ CRT usage 2026-08-01 23:11:10 -04:00
ed 09dde54030 Finished(Controller Input): Reading Joypad State 2026-07-31 15:15:51 -04:00
ed 315e1b2c5e Fix(lua atom tape dsl): Bad-hardcode for source file line-table mapping in dwarf injection pass. 2026-07-31 14:28:50 -04:00
ed 02658d3609 Prepare for hello joypad! 2026-07-28 00:35:02 -04:00
ed dbc459b7e0 gte_hello -> hello_gte. gte is done, moving on to controller! 2026-07-28 00:17:23 -04:00
ed a704341fc6 Testing out the metaprogram with some optimization, need to remove some hardcoding later.. 2026-07-27 23:35:03 -04:00
ed 7421b32fd7 redundant nop reduction 2026-07-27 22:49:41 -04:00
ed e2eb74be19 Remove gte component result contracts (was a bad bodge in, for a later directive thats TODO) 2026-07-27 22:49:26 -04:00
ed 338f1fe46e Better reports from dsl metaprogram 2026-07-27 10:06:23 -04:00
42 changed files with 4239 additions and 1582 deletions
+3
View File
@@ -17,3 +17,6 @@ toolchain/PSn00bSDK
.vscode/settings.json .vscode/settings.json
toolchain/lfs toolchain/lfs
toolchain/lpeg toolchain/lpeg
scratch
toolchain/libpsn00b
+36 -36
View File
@@ -74,41 +74,7 @@
] ]
}, },
{ {
"name": "Debug: Hello GTE Psy-Q!", "name": "Debug: Hello GTE!",
"type": "gdb",
"request": "attach",
"target": "localhost:3333",
"remote": true,
"cwd": "${workspaceRoot}/build",
"valuesFormatting": "parseText",
"registerLimit": "1-32",
"frameFilters": false,
"showDevDebugOutput": false,
"printCalls": false,
"stopAtConnect": true,
"gdbpath": "gdb-multiarch",
"windows": {
"gdbpath": "gdb-multiarch.exe"
},
"osx": {
"gdbpath": "gdb"
},
"executable": "${workspaceRoot}/build/hello_gte.elf",
"setupCommands": [
{ "text": "set mi-async off" },
{ "text": "set remotetimeout 0" },
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
{ "text": "set logging redirect on" }
],
"autorun": [
"monitor reset shellhalt",
"load hello_gte.elf",
"tbreak main",
"continue"
]
},
{
"name": "Debug: Hello GTE Psy-Q! (atoms debug — DWARF-injected)",
"type": "gdb", "type": "gdb",
"request": "attach", "request": "attach",
"target": "localhost:3333", "target": "localhost:3333",
@@ -138,7 +104,41 @@
"monitor reset shellhalt", "monitor reset shellhalt",
"load build/hello_gte.dwarf-injected.elf", "load build/hello_gte.dwarf-injected.elf",
"source scripts/gdb/gdb_tape_atoms.gdb", "source scripts/gdb/gdb_tape_atoms.gdb",
"source build/gen/hello_gte.gdbinit", "tbreak main",
"continue"
]
},
{
"name": "Debug: Hello Joypad!",
"type": "gdb",
"request": "attach",
"target": "localhost:3333",
"remote": true,
"cwd": "${workspaceRoot}",
"valuesFormatting": "parseText",
"registerLimit": "1-32",
"frameFilters": false,
"showDevDebugOutput": false,
"printCalls": false,
"stopAtConnect": true,
"gdbpath": "gdb-multiarch",
"windows": {
"gdbpath": "gdb-multiarch.exe"
},
"osx": {
"gdbpath": "gdb"
},
"executable": "${workspaceRoot}/build/hello_joypad.dwarf-injected.elf",
"setupCommands": [
{ "text": "set mi-async off" },
{ "text": "set remotetimeout 0" },
{ "text": "set logging file build/gen/hello_joypad.gdb.log" },
{ "text": "set logging redirect on" }
],
"autorun": [
"monitor reset shellhalt",
"load build/hello_joypad.dwarf-injected.elf",
"source scripts/gdb/gdb_tape_atoms.gdb",
"tbreak main", "tbreak main",
"continue" "continue"
] ]
+11 -10
View File
@@ -135,16 +135,17 @@ enum { false = 0, true = 1, true_overflow, };
typedef void Proc_(VoidFn) (void); typedef void Proc_(VoidFn) (void);
#define kilo(n) (C_(U4, n) << 10) #define kilo(n) (C_(U4, n) << 10)
#define mega(n) (C_(U4, n) << 20) #define mega(n) (C_(U4, n) << 20)
#define giga(n) (C_(U4, n) << 30) #define giga(n) (C_(U4, n) << 30)
#define tera(n) (C_(U4, n) << 40) #define tera(n) (C_(U4, n) << 40)
#define null C_(U4, 0)
#define nullptr C_(void*, 0) #define null C_(U4, 0)
#define O_(type, field) (C_(U4, & C_(type*,0)->field)) #define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OT_(field) O_(typeof_ptr(& field), filed)) #define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
#define S_(data) C_(U4, sizeof(data)) #define OT_(field) O_(typeof_ptr(& field), filed))
#define S_(data) C_(U4, sizeof(data))
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b)) #define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b)) #define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
+15 -23
View File
@@ -50,17 +50,13 @@
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__)) #define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines // Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
/* reg_str(n) — Stringify an integer register id into the GCC asm /* reg_str(n) — Stringify an integer register id into the GCC asm string form (e.g. 12 → "$12").
* string form (e.g. 12 → "$12"). Use this anywhere GCC's parser * Use this anywhere GCC's parser expects a literal string identifying a register: clobber lists,
* expects a literal string identifying a register: clobber lists, * asm templates, etc. The two-level macro is the standard preprocessor idiom for forcing one level of expansion before stringify —
* asm templates, etc. The two-level macro is the standard preprocessor * without it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead of expanding `R_T4` to its value first.
* idiom for forcing one level of expansion before stringify — without
* it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead
* of expanding `R_T4` to its value first.
* *
* For declaring a register variable bound to a specific GPR, use the * For declaring a register variable bound to a specific GPR, use the `rgcc(n)` bundle from gcc_asm.h instead —
* `rgcc(n)` bundle from gcc_asm.h instead — it adds the `__asm__()` * it adds the `__asm__()` qualifier around the string.
* qualifier around the string.
* *
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose * register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled * register V3_S2* p0 rgcc(R_T4) = ...; // bundled
@@ -85,21 +81,19 @@
* - The string "$12" is derived from it via reg_str, so they cannot drift apart. * - The string "$12" is derived from it via reg_str, so they cannot drift apart.
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise. * - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
* *
* tmpl defined in dsl.h (the token-paste glue). * tmpl defined in dsl.h (token-paste glue).
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific. * rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
* Anyone porting to a different compiler's asm dialect overrides rgcc, * Anyone porting to a different compiler's asm dialect overrides rgcc,
* and the integer→string derivation in rlit can be retargeted in one place. * and the integer→string derivation in rlit can be retargeted in one place.
* *
* For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`. * For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`.
* ------------------------------------------------------------------------ */ * ------------------------------------------------------------------------ */
#define rgcc(n) __asm__(rlit(n)) #define rgcc(n) __asm__(rlit(n))
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used /* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used by the placeholder-pun macros
* by the placeholder-pun macros (the .word bodies are fully baked * (the .word bodies are fully baked at compile time and have no runtime operand references),
* at compile time and have no runtime operand references), but kept * but kept here for completeness in case a future asm template needs to refer to a runtime input by position.
* here for completeness in case a future asm template needs to refer * Mirror of rgcc but produces "%N" instead of "$N". */
* to a runtime input by position. Mirror of rgcc but produces "%N"
* instead of "$N". */
#define rgcc_ref_(n) "%" #n #define rgcc_ref_(n) "%" #n
#define rgcc_ref(n) rgcc_ref_(n) #define rgcc_ref(n) rgcc_ref_(n)
@@ -147,11 +141,9 @@
9, 8, 7, 6, 5, 4, 3, 2, 1, 0)) 9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
/* --- 2. String Concatenation Helpers --- * /* --- 2. String Concatenation Helpers --- *
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's * NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's asm-parser rejects `%cN` in this position with "invalid use of '%c'".
* asm-parser rejects `%cN` in this position with "invalid use of '%c'". * The `%cN` form is for printing *character* constants; for arbitrary integer immediates (the only kind `"i"(...)` produces),
* The `%cN` form is for printing *character* constants; for arbitrary * the plain `%N` form is the right one. Both expand to the bare immediate.
* integer immediates (the only kind `"i"(...)` produces), the plain
* `%N` form is the right one. Both expand to the bare immediate.
*/ */
#define GCC_ASM_W1 "%0" #define GCC_ASM_W1 "%0"
#define GCC_ASM_W2 GCC_ASM_W1 ", %1" #define GCC_ASM_W2 GCC_ASM_W1 ", %1"
+10 -15
View File
@@ -98,11 +98,11 @@ WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
#define mac_gte_store_f3_post_rtpt(...) \ #define mac_gte_store_f3(...) \
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \ gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \ , gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)) , gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
WORD_COUNT(mac_gte_store_f3_post_rtpt, 3) WORD_COUNT(mac_gte_store_f3, 3)
#define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \ #define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \ mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
@@ -115,25 +115,20 @@ WORD_COUNT(mac_format_g4_color, 12)
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2. * G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). * PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 * MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the * (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
* three registers aligned with v0/v1/v2 you must store before RTPS). #define mac_gte_store_g4_p012(...) \
* The macro name declares the pipeline position; check #6 (GTE state-
* machine validation) verifies the call site matches the declaration. */
#define mac_gte_store_g4_p012_post_rtpt_pre_rtps(...) \
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \ gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \ , gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)) , gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3) WORD_COUNT(mac_gte_store_g4_p012, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot. /* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its * PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
* single-vertex result to SXY2; SXY0 still holds v0.screen from the * SXY0 still holds v0.screen from the earlier RTPT.
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
* prevents).
*/ */
#define mac_gte_store_g4_p3_post_rtps(...) \ #define mac_gte_store_g4_p3(...) \
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3))
WORD_COUNT(mac_gte_store_g4_p3_post_rtps, 1) WORD_COUNT(mac_gte_store_g4_p3, 1)
+291 -243
View File
@@ -39,8 +39,8 @@
/* ============================================================================ /* ============================================================================
* Hardware MMIO Addresses * Hardware MMIO Addresses
* ============================================================================ * ============================================================================
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2 * PSX GPU has two 32-bit ports in the I/O register region at KSEG2 0x1F800000+.
* 0x1F800000+. GP0 (offset 0x10) is the data port (commands + params). * GP0 (offset 0x10) is the data port (commands + params).
* GP1 (offset 0x14) is the control port (status, ctrl writes). * GP1 (offset 0x14) is the control port (status, ctrl writes).
* ============================================================================ */ * ============================================================================ */
/* IO base address (KSEG2 0x1F800000+ for the I/O register region). /* IO base address (KSEG2 0x1F800000+ for the I/O register region).
@@ -49,18 +49,18 @@
* `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word). * `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word).
* Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */ * Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */
enum { enum {
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */ IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */ IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros /* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
* that pin a register to IO_BASE_ADDR and access ports via offsets: * that pin a register to IO_BASE_ADDR and access ports via offsets:
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0 * sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */ * sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
GPIO_PORT0_OFFSET = 0x1810, GPIO_PORT0_OFFSET = 0x1810,
GPIO_PORT1_OFFSET = 0x1814, GPIO_PORT1_OFFSET = 0x1814,
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET, HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET, HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
}; };
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR) #define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
@@ -73,66 +73,64 @@ enum {
* GP0 command byte constants + Layer 1 (GPU bitfield shifts) * GP0 command byte constants + Layer 1 (GPU bitfield shifts)
* ============================================================================ * ============================================================================
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only. * 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
* The layer-1 bitfield-layout constants live in the same enum block so the encoder can reference them by name.
* NO macro body past this point uses a raw shift or raw mask. * NO macro body past this point uses a raw shift or raw mask.
* Every shift/width/mask is named here, named once.
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h. * Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp0_cmd_Nop = 0x00, gp0_cmd_Nop = 0x00,
/* Cache management */ /* Cache management */
gp0_cmd_ClearCache = 0x01, gp0_cmd_ClearCache = 0x01,
gp0_cmd_FillVram = 0x02, gp0_cmd_FillVram = 0x02,
gp0_cmd_CopyVram = 0x80, gp0_cmd_CopyVram = 0x80,
gp0_cmd_CopyVramChained = 0x81, gp0_cmd_CopyVramChained = 0x81,
gp0_cmd_ReadVram = 0xC0, gp0_cmd_ReadVram = 0xC0,
/* Polygons */ /* Polygons */
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */ gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */ gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */ gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */ gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */ gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */ gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */ gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */ gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
/* Lines */ /* Lines */
gp0_cmd_line_f2 = 0x40, gp0_cmd_line_f2 = 0x40,
gp0_cmd_line_g2 = 0x50, gp0_cmd_line_g2 = 0x50,
/* Sprites + Tiles + Rects */ /* Sprites + Tiles + Rects */
gp0_cmd_sprt_1 = 0x64, gp0_cmd_sprt_1 = 0x64,
gp0_cmd_sprt_8 = 0x74, gp0_cmd_sprt_8 = 0x74,
gp0_cmd_sprt_16 = 0x7C, gp0_cmd_sprt_16 = 0x7C,
gp0_cmd_tile_1 = 0x60, gp0_cmd_tile_1 = 0x60,
gp0_cmd_tile_8 = 0x68, gp0_cmd_tile_8 = 0x68,
gp0_cmd_tile_16 = 0x70, gp0_cmd_tile_16 = 0x70,
/* State setters (not drawing primitives; set render context). */ /* State setters (not drawing primitives; set render context). */
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */ gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
gp0_cmd_SetTextureWindow = 0xE2, gp0_cmd_SetTextureWindow = 0xE2,
gp0_cmd_SetDrawArea_TopLeft = 0xE3, gp0_cmd_SetDrawArea_TopLeft = 0xE3,
gp0_cmd_SetDrawArea_BotRight = 0xE4, gp0_cmd_SetDrawArea_BotRight = 0xE4,
gp0_cmd_SetDrawOffset = 0xE5, gp0_cmd_SetDrawOffset = 0xE5,
gp0_cmd_SetMaskBit = 0xE6, gp0_cmd_SetMaskBit = 0xE6,
/* bitfield shifts / widths / masks ---- /* bitfield shifts / widths / masks ----
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */ * Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
gp0_cmd_shift = 24, gp0_cmd_shift = 24,
gp0_cmd_width = 8, gp0_cmd_width = 8,
gp0_cmd_mask = 0xFF, gp0_cmd_mask = 0xFF,
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.): /* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
* bits 31..24 = command byte * bits 31..24 = command byte
* bits 23..16 = BLUE * bits 23..16 = BLUE
* bits 15..08 = GREEN * bits 15..08 = GREEN
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */ * bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF, gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF, gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF, gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF, gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
}; };
/* ============================================================================ /* ============================================================================
@@ -171,10 +169,13 @@ enum {
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b)) #define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
/* Cache management — bare-cmd words (no color/range payload). */ /* Cache management — bare-cmd words (no color/range payload). */
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache) #define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram) #define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram) #define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram) #define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
/* NOP — bare-cmd word (no effect; used as DR_ENV padding). */
#define gp0_word_nop() enc_gp0_cmd_word(gp0_cmd_Nop)
/* ============================================================================ /* ============================================================================
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts) * GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
@@ -184,58 +185,57 @@ enum {
* (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`). * (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`).
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp1_cmd_Reset = 0x00, gp1_cmd_Reset = 0x00,
gp1_cmd_ResetCmdBuffer = 0x01, gp1_cmd_ResetCmdBuffer = 0x01,
gp1_cmd_AcknowledgeIRQ = 0x02, gp1_cmd_AcknowledgeIRQ = 0x02,
gp1_cmd_DisplayEnable = 0x03, gp1_cmd_DisplayEnable = 0x03,
gp1_cmd_DMADirection = 0x04, gp1_cmd_DMADirection = 0x04,
gp1_cmd_StartDisplayArea = 0x05, gp1_cmd_StartDisplayArea = 0x05,
gp1_cmd_HorizontalDisplayRange = 0x06, gp1_cmd_HorizontalDisplayRange = 0x06,
gp1_cmd_VerticalDisplayRange = 0x07, gp1_cmd_VerticalDisplayRange = 0x07,
gp1_cmd_DisplayMode = 0x08, gp1_cmd_DisplayMode = 0x08,
/* Note: GP1 only has commands 0x00..0x08. /* Note: GP1 only has commands 0x00..0x08.
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit) * The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
* live in the GP0 enum as * 0xE1..0xE6. * live in the GP0 enum as * 0xE1..0xE6.
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */ * DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode"). /* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
* Bit positions match the encoder shifts below; values are the * Bit positions match the encoder shifts below; values are the *payload* bits only (cmd byte is OR'd in by enc_gp1_disp_mode_word). */
* *payload* bits only (the cmd byte is OR'd in by enc_gp1_disp_mode_word). */ gp1_disp_HRes_256 = 0x0,
gp1_disp_HRes_256 = 0x0, gp1_disp_HRes_320 = 0x1,
gp1_disp_HRes_320 = 0x1, gp1_disp_HRes_512 = 0x2,
gp1_disp_HRes_512 = 0x2, gp1_disp_HRes_640 = 0x3,
gp1_disp_HRes_640 = 0x3, gp1_disp_VRes_240 = 0x0,
gp1_disp_VRes_240 = 0x0, gp1_disp_VRes_480 = 0x1,
gp1_disp_VRes_480 = 0x1, gp1_disp_Color15 = 0x0,
gp1_disp_Color15 = 0x0, gp1_disp_Color24 = 0x1,
gp1_disp_Color24 = 0x1, gp1_disp_VInterlace = 0x1,
gp1_disp_VInterlace = 0x1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */ /* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3, gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1, gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1, gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1, gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */ /* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF, gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF, gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */ /* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF, gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF, gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y /* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
* (10-bit signed — caller pre-signs and masks with the named mask) */ * (10-bit signed — caller pre-signs and masks with the named mask) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF, gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF, gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
}; };
/* ---- Layer 1.5: GP1 per-field encoders ---- */ /* ---- Layer 1.5: GP1 per-field encoders ---- */
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift) #define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift) #define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift) #define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask << gp1_disp_interlace_shift) #define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift) #define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift) #define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
@@ -255,6 +255,11 @@ enum {
#define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y)) #define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
/* ---- Layer 3: GP1 semantic word builders ---- */ /* ---- Layer 3: GP1 semantic word builders ---- */
#define gp1_word_Reset() enc_gp0_cmd_word(gp1_cmd_Reset)
#define gp1_word_ResetCmdBuffer() enc_gp0_cmd_word(gp1_cmd_ResetCmdBuffer)
#define gp1_word_AcknowledgeIRQ() enc_gp0_cmd_word(gp1_cmd_AcknowledgeIRQ)
#define gp1_word_StartDisplayArea() enc_gp0_cmd_word(gp1_cmd_StartDisplayArea)
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1)) #define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
#define gp1_word_display_disable() gp1_word_display_enable(0) #define gp1_word_display_disable() gp1_word_display_enable(0)
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0) #define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
@@ -279,31 +284,36 @@ enum {
#define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable) #define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable)
#define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1) #define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1)
#define gp1_word_DisplayOn() gp1_word_display_enable(0)
#define gp1_word_DisplayOff() gp1_word_display_enable(1)
/* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */ /* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */
enum { enum {
gp1_dma_dir_Off = 0, gp1_dma_dir_Off = 0,
gp1_dma_dir_FIFO = 1, gp1_dma_dir_FIFO = 1,
gp1_dma_dir_CPU_to_GPU = 2, gp1_dma_dir_CPU_to_GPU = 2,
gp1_dma_dir_GPUREAD_to_CPU = 3, gp1_dma_dir_GPUREAD_to_CPU = 3,
}; };
#define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3)) #define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3))
#define gp1_word_dma_to_gpu() gp1_word_dma_direction(gp1_dma_dir_CPU_to_GPU)
#define gp1_word_dma_read_cpu() gp1_word_dma_direction(gp1_dma_dir_GPUREAD_to_CPU)
/* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */ /* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */
/* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */ /* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */
enum { enum {
/* NTSC horizontal range: X1=608, X2=3168 */ /* NTSC horizontal range: X1=608, X2=3168 */
gp1_hrange_NTSC_x1 = 0x260, gp1_hrange_NTSC_x1 = 0x260,
gp1_hrange_NTSC_x2 = 0xC60, gp1_hrange_NTSC_x2 = 0xC60,
/* PAL horizontal range (same as NTSC for most CRTs) */ /* PAL horizontal range (same as NTSC for most CRTs) */
gp1_hrange_PAL_x1 = 0x260, gp1_hrange_PAL_x1 = 0x260,
gp1_hrange_PAL_x2 = 0xC60, gp1_hrange_PAL_x2 = 0xC60,
/* NTSC vertical range: Y1=24, Y2=264 */ /* NTSC vertical range: Y1=24, Y2=264 */
gp1_vrange_NTSC_y1 = 24, gp1_vrange_NTSC_y1 = 24,
gp1_vrange_NTSC_y2 = 264, gp1_vrange_NTSC_y2 = 264,
/* PAL vertical range: Y1=24, Y2=504 */ /* PAL vertical range: Y1=24, Y2=504 */
gp1_vrange_PAL_y1 = 24, gp1_vrange_PAL_y1 = 24,
gp1_vrange_PAL_y2 = 504, gp1_vrange_PAL_y2 = 504,
}; };
#define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2) #define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2)
@@ -314,14 +324,49 @@ enum {
/* ---- Draw-mode setting (TPage / draw-area allowance) ---- */ /* ---- Draw-mode setting (TPage / draw-area allowance) ---- */
/* The "drawing enabled" word is the standard post-init state. */ /* The "drawing enabled" word is the standard post-init state. */
enum { enum {
gp0_DrawMode_DrawToDispBit = 10, /* Per psx-spx, the standard 0xE1 layout has dfe at bit 10. But libpsyx's PutDrawEnv
* uses bit 19 (in the "unused" 14-23 range) for dfe in the DR_ENV code[0] — and the
* PSX hardware honors bit 19 in the DR_ENV context (not bit 10). So we need a
* separate bit definition for the DR_ENV-specific DrawMode. */
gp0_DrawMode_DrawToDispBit = 10, // standard psx-spx bit 10 (dfe)
gp0_DrawMode_DR_ENV_DrawToDispBit = 19, // libpsyx DR_ENV code[0] (dfe in DR_ENV context)
gp0_DrawMode_DR_ENV_isbgBit = 19, // libpsyx uses bit 19 for isbg too
}; };
#define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit)) #define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit))
/* ---- DrawArea pre-baked at origin (0,0) and full screen (320x240) ---- */ /* DR_ENV-specific DrawMode variants (libpsyx SetDrawEnv layout).
* The DR_ENV is a 16-word packet emitted at boot by gp_screen_init's ac_put_draw_env_demo
* atom component. Within the DR_ENV, the 0xE1 command is reused in three different bit
* configurations:
* code[0] = `gp0_word_draw_mode_drawing_allowed` (dfe=1; standard post-init state)
* code[6] = `gp0_word_dr_env_bg_color_cmd(isbg, r, g, b)` (initial-bg-color path)
* code[7] = `gp0_word_dr_env_draw_mode(isbg)` (isbg-flag path)
* Bits 0-23 of the 0xE1 word are the payload; bits 24-31 are the cmd byte (0xE1). */
#define gp0_word_dr_env_bg_color_cmd(isbg, r, g, b) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
#define gp0_word_dr_env_draw_mode(isbg) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0))
/* State-setter bare-cmd words (no immediate payload; the GPU uses the current state machine already programmed). */
#define gp0_word_set_texture_window() enc_gp0_cmd_word(gp0_cmd_SetTextureWindow)
#define gp0_word_set_draw_offset() enc_gp0_cmd_word(gp0_cmd_SetDrawOffset)
#define gp0_word_set_mask_bit() enc_gp0_cmd_word(gp0_cmd_SetMaskBit)
/* DR_ENV code[5] Mask (0xE6 cmd + isbg bit). The isbg bit is set so the GPU knows the auto-clear path is active (paired with code[6] + code[7]). */
#define gp0_word_dr_env_mask() (gp0_word_set_mask_bit() | gp0_dr_env_isbg_bit)
/* DR_ENV pre-baked constants (libpsyx PutDrawEnv layout).
* DR_ENV is a 16-word packet: tag = (length << 24) | addr, where length = 15 (15 code words follow) and addr = 0 (chain to nothing). */
enum {
PolyTag_len_bits = 8,
PolyTag_addr_bits = 24,
gp0_dr_env_tag = (15 << 24) | 0x00FFFFFF,
gp0_dr_env_isbg_bit = (1 << gp0_DrawMode_DR_ENV_isbgBit),
};
/* ---- DrawArea at origin (0,0) and full screen (320x240) ---- */
#define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0) #define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0)
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(320, 240) #define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(319, 239)
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(640, 480) #define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(639, 479)
#pragma endregion GPU Ports & Commands #pragma endregion GPU Ports & Commands
@@ -332,9 +377,9 @@ enum {
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width). * Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
* ============================================================================ */ * ============================================================================ */
enum { enum {
gp1_Status_BitReady = 31, gp1_Status_BitReady = 31,
gp1_Status_BitSendingDMA = 25, gp1_Status_BitSendingDMA = 25,
gp1_Status_DMABlockSizeShift = 0, gp1_Status_DMABlockSizeShift = 0,
}; };
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1) #define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
@@ -360,10 +405,10 @@ typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
#define rgb8(r,g,b) ((RGB8){r,g,b}) #define rgb8(r,g,b) ((RGB8){r,g,b})
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */ /* ---------- PolyTag (the OT-link header; 1 word) ---------- */
enum { // enum {
PolyTag_len_bits = 8, // PolyTag_len_bits = 8,
PolyTag_addr_bits = 24, // PolyTag_addr_bits = 24,
}; // };
typedef Struct_(PolyTag) { typedef Struct_(PolyTag) {
union { union {
U4 code; U4 code;
@@ -387,95 +432,95 @@ typedef Struct_(PolyTag) {
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */ /* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
typedef Struct_(Poly_F3) { typedef Struct_(Poly_F3) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
union { union {
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; }; struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
A3_V2_S2 points; A3_V2_S2 points;
}; };
}; };
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */ /* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
typedef Struct_(Poly_F4) { typedef Struct_(Poly_F4) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
union { union {
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; }; struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
A4_V2_S2 points; A4_V2_S2 points;
}; };
}; };
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */ /* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
typedef Struct_(Poly_G3) { typedef Struct_(Poly_G3) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; V2_S2 p2;
}; };
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */ /* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
typedef Struct_(Poly_G4) { typedef Struct_(Poly_G4) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; RGB8 c3; B1 pad3; V2_S2 p2; RGB8 c3; B1 pad3;
V2_S2 p3; V2_S2 p3;
}; };
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */ /* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */ /* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
typedef Struct_(Poly_FT3) { typedef Struct_(Poly_FT3) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 p0; U1 u0; U1 v0; V2_S2 p0; U1 u0; U1 v0;
V2_S2 p1; U1 u1; U1 v1; V2_S2 p1; U1 u1; U1 v1;
V2_S2 p2; U1 u2; U1 v2; V2_S2 p2; U1 u2; U1 v2;
}; };
/* ---------- Poly_FT4 (Flat Textured Quad) ---------- */ /* ---------- Poly_FT4 (Flat Textured Quad) ---------- */
typedef Struct_(Poly_FT4) { typedef Struct_(Poly_FT4) {
U4 tag; U4 tag;
RGB8 color; RGB8 color;
B1 code; B1 code;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 p0; U1 u0; U1 v0; V2_S2 p0; U1 u0; U1 v0;
V2_S2 p1; U1 u1; U1 v1; V2_S2 p1; U1 u1; U1 v1;
V2_S2 p2; U1 u2; U1 v2; V2_S2 p2; U1 u2; U1 v2;
V2_S2 p3; U1 u3; U1 v3; V2_S2 p3; U1 u3; U1 v3;
}; };
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */ /* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
typedef Struct_(Poly_GT3) { typedef Struct_(Poly_GT3) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; V2_S2 p2;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 tp0; U1 u0; U1 v0; V2_S2 tp0; U1 u0; U1 v0;
V2_S2 tp1; U1 u1; U1 v1; V2_S2 tp1; U1 u1; U1 v1;
V2_S2 tp2; U1 u2; U1 v2; V2_S2 tp2; U1 u2; U1 v2;
}; };
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */ /* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
typedef Struct_(Poly_GT4) { typedef Struct_(Poly_GT4) {
U4 tag; RGB8 c0; B1 code; U4 tag; RGB8 c0; B1 code;
V2_S2 p0; RGB8 c1; B1 pad1; V2_S2 p0; RGB8 c1; B1 pad1;
V2_S2 p1; RGB8 c2; B1 pad2; V2_S2 p1; RGB8 c2; B1 pad2;
V2_S2 p2; RGB8 c3; B1 pad3; V2_S2 p2; RGB8 c3; B1 pad3;
V2_S2 p3; V2_S2 p3;
U4 tpage; U4 tpage;
U4 clut; U4 clut;
V2_S2 tp0; U1 u0; U1 v0; V2_S2 tp0; U1 u0; U1 v0;
V2_S2 tp1; U1 u1; U1 v1; V2_S2 tp1; U1 u1; U1 v1;
V2_S2 tp2; U1 u2; U1 v2; V2_S2 tp2; U1 u2; U1 v2;
V2_S2 tp3; U1 u3; U1 v3; V2_S2 tp3; U1 u3; U1 v3;
}; };
/* ---------- Primitive setters (C-level) ---------- /* ---------- Primitive setters (C-level) ----------
@@ -510,26 +555,29 @@ typedef Struct_(Poly_GT4) {
* bits 12..31 = reserved (zero) * bits 12..31 = reserved (zero)
* ============================================================================ */ * ============================================================================ */
enum { enum {
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */ /* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF, gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1, gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3, gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3, gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1, gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1, gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1, gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
/* TPage color-depth payload values (NOT bit positions — these go in /* TPage color-depth payload values (NOT bit positions — these go in
* the 2-bit field at gp0_tpage_color_depth_shift). */ * the 2-bit field at gp0_tpage_color_depth_shift). */
gp0_tpage_color_4bpp = 0x0, gp0_tpage_color_4bpp = 0x0,
gp0_tpage_color_8bpp = 0x1, gp0_tpage_color_8bpp = 0x1,
gp0_tpage_color_16bpp = 0x2, gp0_tpage_color_16bpp = 0x2,
/* TPage semi-transparency mode payload values (NOT bit positions). */ /* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
gp0_tpage_semi_trans_none = 0x0, gp0_tpage_default = 10,
gp0_tpage_semi_trans_alpha = 0x1,
gp0_tpage_semi_trans_add = 0x2, /* TPage semi-transparency mode payload values (NOT bit positions). */
gp0_tpage_semi_trans_sub = 0x3, gp0_tpage_semi_trans_none = 0x0,
gp0_tpage_semi_trans_alpha = 0x1,
gp0_tpage_semi_trans_add = 0x2,
gp0_tpage_semi_trans_sub = 0x3,
}; };
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */ /* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
@@ -543,19 +591,19 @@ enum {
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */ /* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \ #define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
(enc_gp0_tpage_x(x) \ (enc_gp0_tpage_x(x) \
| enc_gp0_tpage_y(y) \ | enc_gp0_tpage_y(y) \
| enc_gp0_tpage_semi_trans(semi_trans) \ | enc_gp0_tpage_semi_trans(semi_trans) \
| enc_gp0_tpage_color_depth(color_depth) \ | enc_gp0_tpage_color_depth(color_depth) \
| enc_gp0_tpage_dither(dither) \ | enc_gp0_tpage_dither(dither) \
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \ | enc_gp0_tpage_draw_to_disp(draw_to_disp) \
| enc_gp0_tpage_tex_disable(tex_disable)) | enc_gp0_tpage_tex_disable(tex_disable))
typedef Struct_(TexturePage) { U4 raw; }; typedef Struct_(TexturePage) { U4 raw; };
/* ---- Layer 3: TPage semantic word builder ---- */ /* ---- Layer 3: TPage semantic word builder ---- */
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \ #define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable)) enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
#pragma endregion TPage #pragma endregion TPage
#pragma region CLUT #pragma region CLUT
@@ -569,12 +617,12 @@ typedef Struct_(TexturePage) { U4 raw; };
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load) * bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
* ============================================================================ */ * ============================================================================ */
enum { enum {
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */ /* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F, gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF, gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */ /* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
gp0_clut_cmd_Load4bpp = 0x20, gp0_clut_cmd_Load4bpp = 0x20,
gp0_clut_cmd_Load8bpp = 0x25, gp0_clut_cmd_Load8bpp = 0x25,
}; };
/* ---- Layer 1.5: CLUT per-field encoders ---- */ /* ---- Layer 1.5: CLUT per-field encoders ---- */
@@ -615,26 +663,26 @@ enum {
* Stoppped for now at the struct + enum level. * Stoppped for now at the struct + enum level.
* ============================================================================ */ * ============================================================================ */
enum { enum {
tim_file_id_magic = 0x10, tim_file_id_magic = 0x10,
tim_type_4bpp = 0x00, tim_type_4bpp = 0x00,
tim_type_8bpp = 0x01, tim_type_8bpp = 0x01,
tim_type_16bpp = 0x02, tim_type_16bpp = 0x02,
tim_type_32bpp = 0x03, tim_type_32bpp = 0x03,
tim_type_mixed = 0x04, tim_type_mixed = 0x04,
tim_flag_has_clut = 0x08, tim_flag_has_clut = 0x08,
}; };
typedef Struct_(TIM_Header) { typedef Struct_(TIM_Header) {
U4 file_id; /* always 0x10 = "TIM" magic */ U4 file_id; /* always 0x10 = "TIM" magic */
U4 version; /* ignored; always 0 */ U4 version; /* ignored; always 0 */
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */ U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
}; };
typedef Struct_(TIM_SectionHeader) { typedef Struct_(TIM_SectionHeader) {
U4 section_length; /* bytes in this section including this header */ U4 section_length; /* bytes in this section including this header */
U2 org_x; /* origin in VRAM */ U2 org_x; /* origin in VRAM */
U2 org_y; U2 org_y;
U2 width; /* width in pixels */ U2 width; /* width in pixels */
U2 height; /* height in pixels */ U2 height; /* height in pixels */
}; };
#pragma endregion TIM File Format #pragma endregion TIM File Format
View File
+82 -126
View File
@@ -17,9 +17,8 @@
* gte_lw_v0_xy(base) (gte + lw + v0 + xy) * gte_lw_v0_xy(base) (gte + lw + v0 + xy)
* load_upper_i (load-upper + immediate, unique verb) * load_upper_i (load-upper + immediate, unique verb)
* *
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are * Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are NOT in this header.
* NOT in this header. They live in the opt-in `gte_vendor_sym.h` for * They are in the opt-in `gte_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
* users who prefer the textbook MIPS assembly mnemonics.
* ============================================================================ */ * ============================================================================ */
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
@@ -34,20 +33,16 @@
* gte.h — Geometry Transformation Engine (COP2) for the PS1 * gte.h — Geometry Transformation Engine (COP2) for the PS1
* ============================================================================ * ============================================================================
* *
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` * Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C.
* constants from C. No GCC inline-assembly string syntax in the code body. * No GCC inline-assembly string syntax in the code body.
* *
* STYLE NOTES * STYLE NOTES
* ----------- * -----------
* - Per-field encoders are named `enc_gte_<field>(value)` and each one * - Per-field encoders are named `enc_gte_<field>(value)` and each one self-masks its argument before shifting.
* self-masks its argument before shifting. Mirrors the `enc_op / enc_rs * Mirrors the `enc_op / enc_rs / enc_rt / ...` family in mips.h.
* / enc_rt / ...` family in mips.h. * - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of the per-field encoders, plus the COP2/CO base.
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of * - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined for the common cases so call sites read like assembly source.
* the per-field encoders, plus the COP2/CO base. * - All register/field values are enums (not `#define`s) so they show up in debugger symbol tables and IDE autocomplete.
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined
* for the common cases so call sites read like assembly source.
* - All register/field values are enums (not `#define`s) so they show up
* in debugger symbol tables and IDE autocomplete.
* *
* SEE ALSO * SEE ALSO
* -------- * --------
@@ -58,8 +53,7 @@
/* --- GTE Data Registers (Coprocessor 2) --- /* --- GTE Data Registers (Coprocessor 2) ---
* Preprocessor-visible integer ids for the COP2 data register file. * Preprocessor-visible integer ids for the COP2 data register file.
* Each enum value is bound to a parallel `_Code` `#define` so the * Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
* Same pattern as the GPR `_Code` set in mips.h. */ * Same pattern as the GPR `_Code` set in mips.h. */
#define C2_VXY0_Code 0 #define C2_VXY0_Code 0
#define C2_VZ0_Code 1 #define C2_VZ0_Code 1
@@ -192,10 +186,8 @@ enum {
/* --- GTE Control Register Indices (for ctc2/cfc2) --- /* --- GTE Control Register Indices (for ctc2/cfc2) ---
* Preprocessor-visible integer ids for the COP2 control register file. * Preprocessor-visible integer ids for the COP2 control register file.
* Each enum value is bound to a parallel `_Code` `#define` so the * Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths). * Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23 are reserved/unused on real hardware, so there's a gap. */
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23
* are reserved/unused on real hardware, so there's a gap. */
#define gte_cr_RT11_Code 0 #define gte_cr_RT11_Code 0
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */ #define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */ #define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
@@ -223,8 +215,9 @@ enum {
#define gte_cr_RFC_Code 27 #define gte_cr_RFC_Code 27
#define gte_cr_GFC_Code 28 #define gte_cr_GFC_Code 28
#define gte_cr_BFC_Code 29 #define gte_cr_BFC_Code 29
#define gte_cr_OFX_Code 30 #define gte_cr_OFX_Code 24
#define gte_cr_OFY_Code 31 #define gte_cr_OFY_Code 25
#define gte_cr_H_Code 26
enum { enum {
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code, gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
@@ -246,21 +239,16 @@ enum { _C2_OPS_ = 0
/* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx). /* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx).
* *
* Spans the 2x2 {From, To} × {Data, Control} register classes that the * Spans the 2x2 {From, To} × {Data, Control} register classes that the GTE exposes:
* GTE exposes:
*
* bit 1 (0x02): register class — 0 = data, 1 = control * bit 1 (0x02): register class — 0 = data, 1 = control
* bit 2 (0x04): direction — 0 = read, 1 = write * bit 2 (0x04): direction — 0 = read, 1 = write
* *
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit * The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
* numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h * (which target the data register file on any coprocessor).
* (which target the data register file on any coprocessor). They are * They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
* re-aliased here so the four-way table reads like the spec mnemonics * and so the encoding lives next to its only consumer (this header).
* (MFC2 / CFC2 / MTC2 / CTC2) and so the encoding lives next to its
* only consumer (this header).
* *
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) * Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
* live in gte_vendor_sym.h. */
enum { _C2_TX_SUBS_ = 0 enum { _C2_TX_SUBS_ = 0
, sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */ , sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */
, sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */ , sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */
@@ -270,11 +258,11 @@ enum { _C2_TX_SUBS_ = 0
/* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd /* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11] * Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2 * - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
* - rt: GPR source/dest * - rt: GPR source/dest
* - rd: COP2 register index (0..31): * - rd: COP2 register index (0..31):
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases) * data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */ * ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd)) #define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
@@ -314,8 +302,8 @@ enum { _C2_TX_SUBS_ = 0
* `swc2` is redundant when we're already inside the `gte_` namespace. * `swc2` is redundant when we're already inside the `gte_` namespace.
* gte_lw rt, base, off → lwc2 rt, off(base) * gte_lw rt, base, off → lwc2 rt, off(base)
* gte_sw rt, base, off → swc2 rt, off(base) * gte_sw rt, base, off → swc2 rt, off(base)
* For the typical user-facing vector-level load (xy + z as two * For the typical user-facing vector-level load (xy + z as two instructions),
* instructions), use the higher-level `gte_load_vN` macros below. */ * use the higher-level `gte_load_vN` macros below. */
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off) #define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off) #define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
@@ -323,13 +311,12 @@ enum { _C2_TX_SUBS_ = 0
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO). * Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
* The lower 25 bits are the GTE-specific command payload. * The lower 25 bits are the GTE-specific command payload.
* *
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` * The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
* pattern in mips.h: each one self-masks and shifts its own field, so a * Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
* caller can build up a GTE command piece by piece (handy for state-driven * (handy for state-driven MVMVA emitters that vary one field at a time).
* MVMVA emitters that vary one field at a time).
* *
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command * `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go.
* word in one go. It just ORs the per-field encoders together. */ * It just ORs the per-field encoders together. */
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25)) #define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
/* Per-field encoders. Each one does (value & mask) << shift on its own. */ /* Per-field encoders. Each one does (value & mask) << shift on its own. */
@@ -359,12 +346,13 @@ enum { _C2_TX_SUBS_ = 0
* Decomposition (per the `enc_gte_<field>` definitions above): * Decomposition (per the `enc_gte_<field>` definitions above):
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>) * gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
* The SF/MX/V/CV/LM fields are all zero in the common cases * The SF / MX / V / CV / LM fields are all zero in the common cases
* (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp), * (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
* so the only varying bits are the `cmd` field. * so the only varying bits are the `cmd` field.
* *
* Naming follows the file's convention: `gte_cmd_*` is the raw 6-bit `cmd` field id, `gte_cmdw_*` * Naming convention:
* is the fully-encoded 32-bit instruction word ready to drop into a `.word` directive. * - `gte_cmd_*` : Raw 6-bit `cmd` field id
* - `gte_cmdw_* : 32-bit instruction word ready to drop into a `.word` directive.
* *
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
* PsyQ-compatibility note (RTPS/RTPT): * PsyQ-compatibility note (RTPS/RTPT):
@@ -380,7 +368,7 @@ enum { _C2_TX_SUBS_ = 0
* `nclip` ends up wrong, and the triangle is culled. * `nclip` ends up wrong, and the triangle is culled.
* *
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years. * So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source. * NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
*/ */
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer)) #define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
@@ -413,20 +401,16 @@ enum { _C2_TX_SUBS_ = 0
/** /**
* @brief Loads a single SVECTOR to GTE vector register V0 * @brief Loads a single SVECTOR to GTE vector register V0
*
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0 * @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`. * (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
* *
* Uses string-style GCC inline asm with `%0` substitution because the * Uses string-style GCC inline asm with `%0` substitution because the base register `r0` is a runtime GPR chosen by the compiler.
* base register `r0` is a runtime GPR chosen by the compiler.
* It cannot be encoded into a static `.word` constant. * It cannot be encoded into a static `.word` constant.
* *
* Usage: * Usage: asm_gte_load_v0(svector_ptr);
* asm_gte_load_v0(svector_ptr);
*/ */
/* lwc2 encoding helpers parameterized on the base GPR. /* lwc2 encoding helpers parameterized on the base GPR.
*
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0 * gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0 * gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1 * gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
@@ -435,8 +419,7 @@ enum { _C2_TX_SUBS_ = 0
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2 * gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
* *
* `base` is the GPR number to bake into the .word constant's `rs` field. * `base` is the GPR number to bake into the .word constant's `rs` field.
* These are pure compile-time integers; the C compiler constant-folds * These are pure compile-time integers; the C compiler constant-folds them into .word directives. */
* them into .word directives. */
enum { enum {
GTE_Z_Offset = 4 GTE_Z_Offset = 4
@@ -459,8 +442,8 @@ enum {
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12 * gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
* *
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in), * Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
* which is exactly the register the .word constants expect. A `"$12"` clobber would conflict with the register-variable binding * which is exactly the register the .word constants expect.
* ("asm specifier for variable conflicts with asm clobber list"), so we omit it. * A `"$12"` clobber would conflict with the register-variable binding ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
* The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile. * The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
* *
* WHICH REGISTER TO PICK * WHICH REGISTER TO PICK
@@ -499,10 +482,8 @@ enum {
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt. /* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
* *
* Loads all three GTE input vectors (6 words) from three separate pointers, * Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register,
* one per GTE vector register, each loaded from its own base GPR. * each loaded from its own base GPR. Caller must bind each `pN` to `bN` via a register variable.
* Caller must bind each `pN` to `bN` via a register variable.
*
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12") * register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13") * register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14") * register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
@@ -521,29 +502,20 @@ enum {
/** /**
* @brief Rotate, Translate and Perspective Triple (23 cycles) * @brief Rotate, Translate and Perspective Triple (23 cycles)
* * @details Performs rotation, translation and perspective calculation of three vertices at once.
* @details Performs rotation, translation and perspective calculation of three * The equation performed is the same as gte_rtps() only repeated three times for each vertex.
* vertices at once. The equation performed is the same as gte_rtps() only * The result of the first vertex is stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then C2_SXY2.
* repeated three times for each vertex. The result of the first vertex is
* stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then
* C2_SXY2.
* *
* Encoder-style emission (no inline-asm strings in the code body): * Encoder-style emission (no inline-asm strings in the code body):
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE * 1. Two `nop` words fill the COP2 pipeline latency — the GTE takes ~8 cycles per perspective divide,
* takes ~8 cycles per perspective divide, and the nops let any * and the nops let any preceding lwc2/swc2 retire before RTPT starts reading its inputs from V0/V1/V2.
* preceding lwc2/swc2 retire before RTPT starts reading its * 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the pre-baked encoders above) —
* inputs from V0/V1/V2. * `0x0280030` decoded as `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields zero
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the * (standard rotation, no scaling, V0 vector, translation vector, no clamp).
* pre-baked encoders above) — `0x0280030` decoded as
* `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields
* zero (standard rotation, no scaling, V0 vector, translation
* vector, no clamp).
* *
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel * Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI)
* ABI) plus the standard "memory" barrier. Does not clobber any COP2 * plus the standard "memory" barrier. Does not clobber any COP2 data/control register —
* data/control register — those have to be saved by the caller if * those have to be saved by the caller if they need to survive across the call (RTPT writes SXY0..2, SZ0..3, OTZ, MAC0..3, IR0..3, etc.).
* they need to survive across the call (RTPT writes SXY0..2, SZ0..3,
* OTZ, MAC0..3, IR0..3, etc.).
*/ */
#define gte_rtpt() \ #define gte_rtpt() \
asm volatile( \ asm volatile( \
@@ -559,32 +531,24 @@ enum {
/** /**
* @brief Normal clipping (8 cycles) * @brief Normal clipping (8 cycles)
* * @details Computes the sign of three screen coordinates (C2_SXY0-2) used for backface culling.
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for * If the value of C2_MAC0 is negative, the coordinates are inverted and thus the triangle is back facing.
* backface culling. If the value of C2_MAC0 is negative, the coordinates are
* inverted and thus the triangle is back facing.
* *
* The following equation is performed when executing this GTE command: * The following equation is performed when executing this GTE command:
*
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1 * MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
*
* Encoder-style emission (no inline-asm strings in the code body): * Encoder-style emission (no inline-asm strings in the code body):
* 1. Two `nop` words fill the COP2 pipeline latency - the GTE * 1. Two `nop` words fill the COP2 pipeline latency
* pipeline takes a few cycles per op, and the nops let any * - the GTE pipeline takes a few cycles per op, and the nops let any preceding
* preceding lwc2/swc2/RTPT retire before NCLIP starts reading * lwc2/swc2/RTPT retire before NCLIP starts reading its inputs from SXY0/SXY1/SXY2.
* its inputs from SXY0/SXY1/SXY2. * 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the pre-baked encoders above)
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the * - `0x01400006` decoded as `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields zero.
* pre-baked encoders above) - `0x01400006` decoded as * NCLIP is spec-clean in the original PsyQ source (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat` quirk),
* `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields * so `gte_cmdw_nclip` does NOT OR in any reserved bits.
* zero. NCLIP is spec-clean in the original PsyQ source
* (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat`
* quirk), so `gte_cmdw_nclip` does NOT OR in any reserved bits.
* *
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel * Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI) plus the standard "memory" barrier.
* ABI) plus the standard "memory" barrier. Does not clobber any COP2 * Does not clobber any COP2 data/control register.
* data/control register - those have to be saved by the caller if * Those have to be saved by the caller if they need to survive across the call (NCLIP writes MAC0 only;
* they need to survive across the call (NCLIP writes MAC0 only; it * it is purely a sign-of-double-product computation on SXY0..2).
* is purely a sign-of-double-product computation on SXY0..2).
*/ */
#define gte_nclip() \ #define gte_nclip() \
asm volatile( \ asm volatile( \
@@ -610,13 +574,10 @@ enum {
"cop2 0x0158002D;") "cop2 0x0158002D;")
/* asm_gte_matrix_set_rotation(r0) /* asm_gte_matrix_set_rotation(r0)
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
* *
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix * Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
* control registers (RT11..RT22, indices 0..4) via ctc2. * The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as
*
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16),
* each holding two packed 16-bit matrix elements. The first 1.5 rows
* of a standard PSX SDK MATRIX struct (where each row is laid out as
* [RT_xx, RT_xy] | [RT_xz, pad] | ...). * [RT_xx, RT_xy] | [RT_xz, pad] | ...).
* *
* Generated MIPS (mirrors the source macro): * Generated MIPS (mirrors the source macro):
@@ -631,27 +592,22 @@ enum {
* ctc2 $13, $3 ; → C2_RT21 * ctc2 $13, $3 ; → C2_RT21
* ctc2 $14, $4 ; → C2_RT22 * ctc2 $14, $4 ; → C2_RT22
* *
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a * Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
* register variable (`rgcc(R_T4)`) for the `lw $12, off(...)` * instructions to read from the right base. The `"r"(r0)` constraint alone doesn't force a specific GPR — it just lets GCC pick one.
* instructions to read from the right base. The `"r"(r0)` constraint * The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field of each lw, so the lw instructions will
* alone doesn't force a specific GPR — it just lets GCC pick one. * only do the right thing if $12 / $13 / $14 hold the matrix base at runtime.
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field
* of each lw, so the lw instructions will only do the right thing
* if $12/$13/$14 hold the matrix base at runtime.
* *
* M3_S2* m = ...; * M3_S2* m = ...;
* register M3_S2* m_in_12 rgcc(R_T4) = m; * register M3_S2* m_in_12 rgcc(R_T4) = m;
* asm_gte_matrix_set_rotation(m_in_12); * asm_gte_matrix_set_rotation(m_in_12);
* *
* We clobber $12/$13/$14 (the ones we use as scratch inside the * We clobber $12/$13/$14 (the ones we use as scratch inside the inline asm)
* inline asm) plus the system clobbers; we don't clobber `r0` because * plus the system clobbers; we don't clobber `r0` because the `rgcc` binding already says "this variable lives in $12".
* the `rgcc` binding already says "this variable lives in $12".
* *
* WARNING: Incomplete by design. The source macro only writes RT11..RT22 * WARNING: Incomplete by design. The source macro only writes RT11..RT22 (5 of 9 rotation elements);
* (5 of 9 rotation elements); RT23 and the entire RT3x row are left * RT23 and the entire RT3x row are left untouched.
* untouched. Real libpsn00b SetRotMatrix writes all 9. Use only when the * Real libpsn00b SetRotMatrix writes all 9. Use only when the GTE's remaining rotation entries are already correct,
* GTE's remaining rotation entries are already correct, or you will * or you will get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
* get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
*/ */
#define asm_gte_matrix_set_rotation(r0) \ #define asm_gte_matrix_set_rotation(r0) \
asm volatile( \ asm volatile( \
View File
+81 -118
View File
@@ -10,10 +10,10 @@
# include "gen/duffle.offsets.h" # include "gen/duffle.offsets.h"
#endif #endif
typedef U4 const MipsCode; typedef U4 const MipsCode; // Underlying type to mips asm words.
typedef Slice_(MipsCode); typedef Slice_(MipsCode);
typedef Slice_MipsCode MipsAtom;
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
#define MipsAtom_(sym) MipsCode sym [] align_(4) = #define MipsAtom_(sym) MipsCode sym [] align_(4) =
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names). // Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
@@ -23,63 +23,78 @@ typedef Slice_MipsCode MipsAtom;
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) = #define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
// Used for components with value-args (e.g., ac_format_f3_color). // Used for components with value-args (e.g., ac_format_f3_color).
// FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body }) // FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
// expands to: // expands to:
// FI_ MipsAtom ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); } // FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); } #define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
// Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h) are included manually by the unity build. // Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h) are included manually by the unity build.
/* Register aliases */ /* Register aliases */
enum { enum {
R_AtomJmp = R_T9 atom_reg, /* debug-visible; tape yield handshake scratch */ R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
R_TapePtr = R_T8 atom_reg, /* The Instruction Stream Pointer */ R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
R_InCursor = R_T4,
R_PrimCursor = R_T7 atom_reg atom_type(U4 *), /* VRAM output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2 *), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
R_VertBase = R_T5 atom_reg atom_type(V3_S2 *), /* Base address of the vertex array */
R_OtBase = R_T6 atom_reg atom_type(U4 *), /* Base address of the Ordering Table */
/* Stringification codes for the GCC inline assembler clobber lists. */ /* Stringification codes for the GCC inline assembler clobber lists. */
#define R_TapePtr_Code R_T8_Code #define R_AtomJmp_Code R_T8_Code
#define R_InCursor_Code R_T4_Code #define R_TapePtr_Code R_T9_Code
#define R_PrimCursor_Code R_T7_Code // R_InCursor = R_T4,
#define R_FaceCursor_Code R_T4_Code // #define R_InCursor_Code R_T4_Code
#define R_VertBase_Code R_T5_Code
#define R_OtBase_Code R_T6_Code // Reserved Registers (Callee-saved):
// - R_T9: Holds the Tape Ptr which we need to increment
// If we hit a wall with register allocations we can clobber V0 & V1 (return values), defering as opt-in by user.
// - R_RA: Not sure??
// Needed by ac_yield but can be used as atom scratch:
// - R_T8: Will be used as the atom jump register.
// All allocatable registers for mips atoms:
R_TScratchVolatile = R_AT, // This one is reserved for psuedo instructions, but you can technically use it.
R_TScratch0 = R_T0,
R_TScratch1 = R_T1,
R_TScratch2 = R_T2,
R_TScratch3 = R_T3,
R_TScratch4 = R_T4,
R_TScratch5 = R_T5,
R_TScratch6 = R_T6,
R_TScratch7 = R_T7,
R_TScratch8 = R_T8,
R_TScratch10 = R_V0,
R_TScratch11 = R_V1,
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
// R_TScratch12 = R_A0,
// R_TScratch13 = R_A1,
// R_TScratch14 = R_A3,
// TODO(Ed): Review S0-S7, they are technically avaialble, we just have to snapshot them at the ABI boundary.
// TODO(Ed): This is technically a waste of cycles for most work? so maybe only do this for expensive atoms on-demand or atom phases.
// TODO(Ed): Sort out the other available registers... (Not sure how much is left avail)
}; };
#pragma region Tape Drive #pragma region Tape Drive
/* --------------------------------------------------------------------------- /* ---------------------------------------------------------------------------
* TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below) * TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below)
* ---------------------------------------------------------------------------*/ * ---------------------------------------------------------------------------*/
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
/* The 'Exit' Atom */ /* The 'Exit' Atom */
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }; atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
//TODO(Ed): Do we backup R_S0-7 here? Have it in a heavier tape run as a opt-in? Same with V0-1 and A0-3?
/* Generalized Tape Engine Runner */ /* Generalized Tape Engine Runner */
NI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile( FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
asm_words( asm_words(
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */ load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */ , add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
, load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */ , call_reg( R_AtomJmp) /* jalr $t9 */
, add_ui_self(R_TapePtr, S_(MipsCode)) /* Advance tape */ , nop /* Branch delay slot */
, call_reg( R_AtomJmp) /* jalr $t9 */
, nop /* Branch delay slot */
, load_word( R_RA, R_SP, 0) /* Restore $ra from stack */
, add_ui_self(R_SP, MipsStackAlignment) /* Deallocate stack space */
) )
asm_rpins, r_use(tp) asm_rpins, r_use(tape_ptr)
asm_clobber: asm_clobber:
rlit(R_AT) rlit(R_AT),
, rlit(R_V0), rlit(R_V1) rlit(R_V0), rlit(R_V1),
, rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3) rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
/* Tell GCC the tape engine owns and destroys the workspace registers */ rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
, rlit(R_PrimCursor), rlit(R_FaceCursor), rlit(R_VertBase), rlit(R_OtBase) clb_mem_drain
, rlit(R_T9)
, clb_mem_drain
); } ); }
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; }; typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
@@ -87,13 +102,17 @@ FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; } FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; } FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
#define tb_emit_(tb, atom) tb_emit(tb, atom)
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; } FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; } FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
#define tb_emit_(atom) tb_emit(& tb, atom)
#define tb_data_(field, data) tb_data(& tb, u4_(data))
FI_ Slice_MipsCode tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Slice_MipsCode){ C_(U4*,tb->ptr), tb->used }; } FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Slice_MipsCode){ C_(U4*,tb.ptr), tb.used }; } FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit)) #define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
#pragma endregion Tape Drive #pragma endregion Tape Drive
@@ -110,6 +129,17 @@ atom_dbg_skip MipsAtomComp_(ac_yield) {
jump_reg( R_AtomJmp), nop, jump_reg( R_AtomJmp), nop,
}; };
enum {
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
#define R_PrimCursor_Code R_T7_Code
#define R_FaceCursor_Code R_T4_Code
#define R_VertBase_Code R_T5_Code
#define R_OtBase_Code R_T6_Code
};
/* Words: 3; Loads 3 S2 indices from the face array */ /* Words: 3; Loads 3 S2 indices from the face array */
atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) { atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) {
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)), load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
@@ -156,7 +186,7 @@ MipsAtomComp_(ac_insert_ot_tag_g4) {
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given /* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
* byte offset. Internal helper used by the *_format_*_color macros. */ * byte offset. Internal helper used by the *_format_*_color macros. */
FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b) FI_ Slice_MipsCode ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, { atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
load_upper_i(R_AT, (cmd) << 8 | (b)), load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)), or_i_self( R_AT, ((g) << 8) | (r)),
@@ -165,12 +195,12 @@ atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED) /* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */ * Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b) FI_ Slice_MipsCode ac_format_f3_color(U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) }) atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) { atom_dbg_skip MipsAtomComp_(ac_gte_store_f3) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)), gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)), gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)), gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
@@ -178,7 +208,7 @@ atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
/* Words: 12; Emits the four (code|color) words of a Poly_G4. /* Words: 12; Emits the four (code|color) words of a Poly_G4.
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */ * Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
FI_ MipsAtom ac_format_g4_color( FI_ Slice_MipsCode ac_format_g4_color(
U1 r0, U1 g0, U1 b0, U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1, U1 r1, U1 g1, U1 b1,
U1 r2, U1 g2, U1 b2, U1 r2, U1 g2, U1 b2,
@@ -193,24 +223,19 @@ MipsAtomComp_Proc_(ac_format_g4_color, {
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
* G4 triangle portion to p0/p1/p2. * G4 triangle portion to p0/p1/p2.
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). * PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 * MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the * (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
* three registers aligned with v0/v1/v2 you must store before RTPS). atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012) {
* The macro name declares the pipeline position; check #6 (GTE state-
* machine validation) verifies the call site matches the declaration. */
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)), gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)), gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)), gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
}; };
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot. /* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its * PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
* single-vertex result to SXY2; SXY0 still holds v0.screen from the * SXY0 still holds v0.screen from the earlier RTPT.
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
* prevents).
*/ */
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) }; atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
#pragma endregion Macro Atom Components #pragma endregion Macro Atom Components
@@ -237,7 +262,7 @@ FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield)); mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
} }
#define mipsatom_from_builder(ab) (MipsAtom){ab.start, ab.used} #define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used}
#pragma endregion Mips Atom Builder #pragma endregion Mips Atom Builder
@@ -291,66 +316,4 @@ internal MipsAtom_(set_gte_world) atom_info(
mac_yield() mac_yield()
}; };
/* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() };
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0),
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
store_word( R_AT, R_T7, 4),
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
add_ui( R_T1, R_0, 10),
shift_lleft_self(R_T1, S_(U4)/2),
add_u_self( R_T1, R_T6),
load_word( R_AT, R_T1, 0),
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
store_word( R_AT, R_T7, 0),
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
or_u_self( R_AT, R_V0),
store_word( R_AT, R_T1, 0),
add_ui(R_T7, R_T7, 20),
mac_yield()
};
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) {
/* Load 3 indices */
load_half_u(R_T0, R_T4, 0),
load_half_u(R_T1, R_T4, 2),
load_half_u(R_T2, R_T4, 4),
/* Load Vertices into GTE */
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
/* Run Math */
nop2, gte_cmdw_rtpt,
nop2, gte_cmdw_nclip,
nop2,
/* Advance Face Cursor and Yield */
add_ui(R_T4, R_T4, 8),
mac_yield()
};
#pragma endregion Baked Mips Atoms #pragma endregion Baked Mips Atoms
+3
View File
@@ -11,6 +11,7 @@ enum {
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset. v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
}; };
typedef Array_(U1, 2);
typedef Array_(U4, 2); typedef Array_(U4, 2);
typedef Array_(S2, 2); typedef Array_(S2, 2);
typedef Array_(S2, 3); typedef Array_(S2, 3);
@@ -22,6 +23,7 @@ typedef S2 A3x3_S2[3][3];
typedef Struct_(Extent2_S2) { S2 width; S2 height; }; typedef Struct_(Extent2_S2) { S2 width; S2 height; };
typedef Struct_(Extent2_S4) { S4 width; S4 height; }; typedef Struct_(Extent2_S4) { S4 width; S4 height; };
typedef Struct_(V2_U1) { U1 x; U1 y; };
typedef Struct_(V2_S2) { S2 x; S2 y; }; typedef Struct_(V2_S2) { S2 x; S2 y; };
typedef Struct_(V2_S4) { S4 x; S4 y; }; typedef Struct_(V2_S4) { S4 x; S4 y; };
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
@@ -37,6 +39,7 @@ typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; }; typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
typedef Array_(V2_S2, 2);
typedef Array_(V2_S2, 3); typedef Array_(V2_S2, 3);
typedef Array_(V2_S2, 4); typedef Array_(V2_S2, 4);
+2 -2
View File
@@ -67,8 +67,8 @@ typedef Slice_(B1);
#define slice_end(slice) ((slice).ptr + (slice).len) #define slice_end(slice) ((slice).ptr + (slice).len)
#define S_slice(s) ((s).len * S_((s).ptr[0])) #define S_slice(s) ((s).len * S_((s).ptr[0]))
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len)) #define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a)) #define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s)) #define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter) #define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
+6 -9
View File
@@ -336,10 +336,10 @@ enum { _BitOffsets = 0
/* Logic Opcodes */ /* Logic Opcodes */
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and) #define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or) #define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor) #define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor) #define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
#define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or) #define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or)
@@ -444,18 +444,15 @@ enum { _BitOffsets = 0
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm)) #define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi). /* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
*
* Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when: * Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when:
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR * - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
* - `imm` is not a compile-time constant and you want predictable * - `imm` is not a compile-time constant and you want predictable 2-word emission without the `__builtin_constant_p` branches.
* 2-word emission without the `__builtin_constant_p` branches.
* *
* The lo16 strategy is still chosen at expansion time on the lo half: * The lo16 strategy is still chosen at expansion time on the lo half:
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0) * lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0)
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern) * lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern)
* *
* For situations where you need to bypass even this choice * For situations where you need to bypass even this choice (e.g. to force a specific encoding for a known discontiguous high/low pair),
* (e.g. to force a specific encoding for a known discontiguous high/low pair),
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below. * see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
* Statement-level (not expression-level): emits its own `asm volatile(...)`. * Statement-level (not expression-level): emits its own `asm volatile(...)`.
*/ */
+73
View File
@@ -0,0 +1,73 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "dsl.h"
#endif
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
* Wire is active-low (0 = pressed).
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */
enum {
Bit_(Pad_Select, 0),
Bit_(Pad_L3, 1),
Bit_(Pad_R3, 2),
Bit_(Pad_Start, 3),
Bit_(Pad_Up, 4),
Bit_(Pad_Right, 5),
Bit_(Pad_Down, 6),
Bit_(Pad_Left, 7),
Bit_(Pad_L2, 8),
Bit_(Pad_R2, 9),
Bit_(Pad_L1, 10),
Bit_(Pad_R1, 11),
Bit_(Pad_Triangle, 12),
Bit_(Pad_Circle, 13),
Bit_(Pad_Cross, 14),
Bit_(Pad_Square, 15),
};
enum {
PadId_Offset = 4,
Pad0 = 0 << PadId_Offset,
Pad1 = 1 << PadId_Offset,
};
#define pad0_(btn_id) (btn_id << Pad0)
#define pad1_(btn_id) (btn_id << Pad1)
/* ============================================================
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
* ============================================================ */
enum {
PAD_BIOS_RAW_SIZE = 0x22,
};
typedef Struct_(PadBiosRaw) {
U1 bytes[PAD_BIOS_RAW_SIZE];
};
typedef Enum_(U4, PadStatus) {
PadStatus_Disconnected,
PadStatus_Digital,
PadStatus_AnalogStick,
PadStatus_AnalogPad,
PadStatus_Unsupported,
PadStatus_Pending,
PadStatus_Invalid,
};
/* PadState — per-port normalized runtime state.
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
* The struct size stays 12 bytes (unchanged from the prior order,
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */
typedef Struct_(PadState) {
PadStatus status; /* offset 0, size 4 (U4) */
U2 buttons; /* offset 4, size 2 */
U1 id; /* offset 6, size 1 */
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
U1 left_y; /* offset 9, size 1 */
U1 right_x; /* offset 10, size 1 */
U1 right_y; /* offset 11, size 1 */
};
View File
+4
View File
@@ -22,6 +22,7 @@ WORD_COUNT(call_reg, 1)
WORD_COUNT(call_addr, 1) WORD_COUNT(call_addr, 1)
WORD_COUNT(branch_le_zero, 1) WORD_COUNT(branch_le_zero, 1)
WORD_COUNT(branch_equal, 1) WORD_COUNT(branch_equal, 1)
WORD_COUNT(branch_ne, 1)
WORD_COUNT(add_ui, 1) WORD_COUNT(add_ui, 1)
WORD_COUNT(set_lt_u, 1) WORD_COUNT(set_lt_u, 1)
WORD_COUNT(set_lt_s, 1) WORD_COUNT(set_lt_s, 1)
@@ -29,7 +30,9 @@ WORD_COUNT(set_lt_si, 1)
WORD_COUNT(set_lt_ui, 1) WORD_COUNT(set_lt_ui, 1)
WORD_COUNT(load_word, 1) WORD_COUNT(load_word, 1)
WORD_COUNT(load_half_u, 1) WORD_COUNT(load_half_u, 1)
WORD_COUNT(load_byte_u, 1)
WORD_COUNT(store_word, 1) WORD_COUNT(store_word, 1)
WORD_COUNT(store_byte, 1)
WORD_COUNT(add_ui_self, 1) WORD_COUNT(add_ui_self, 1)
WORD_COUNT(add_u_self, 1) WORD_COUNT(add_u_self, 1)
WORD_COUNT(add_u, 1) WORD_COUNT(add_u, 1)
@@ -37,6 +40,7 @@ WORD_COUNT(or_i, 1)
WORD_COUNT(or_i_self, 1) WORD_COUNT(or_i_self, 1)
WORD_COUNT(or_u, 1) WORD_COUNT(or_u, 1)
WORD_COUNT(or_u_self, 1) WORD_COUNT(or_u_self, 1)
WORD_COUNT(nor_u, 1)
WORD_COUNT(shift_lleft, 1) WORD_COUNT(shift_lleft, 1)
WORD_COUNT(shift_lleft_self, 1) WORD_COUNT(shift_lleft_self, 1)
WORD_COUNT(shift_lright, 1) WORD_COUNT(shift_lright, 1)
-156
View File
@@ -1,156 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "duffle/gen/duffle.macs.h"
# include "duffle/gen/duffle.offsets.h"
# include "duffle/atom_dsl.h"
# include "duffle/lottes_tape.h"
# include "duffle/word_count.metadata.h"
# include "gen/gte_hello.offsets.h"
# include "hello_gte.h"
#endif
#pragma region MACs (Mips Atom components)
#pragma endregion MACs
#pragma region Baked Atoms
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple,
nop2, gte_cmdw_nclip,
nop2, gte_mv_from_data_r(R_T0, C2_MAC0),
nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
mac_format_g4_color(
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
mac_gte_store_g4_p012_post_rtpt_pre_rtps(),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
nop2, gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3_post_rtps(),
nop2, gte_cmdw_avg_sort_z4,
nop2, gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(),
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursr)
) {
mac_load_tri_indices( R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_gte_store_f3_post_rtpt(),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
/* Insert into Ordering Table Linked List */
mac_insert_ot_tag_f3(),
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
#pragma endregion Baked Atoms
@@ -5,10 +5,10 @@
#pragma region hello_gte_tape #pragma region hello_gte_tape
// --- atom: cube_g4_face (87 words) --- // --- atom: cube_g4_face (77 words) ---
#define _atom_offset_cull_cube_g4_face_exit 48 #define _atom_offset_cull_cube_g4_face_exit 42
#define _atom_offset_bounds_chk_cube_g4_face_exit 12 #define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum { enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit, atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
@@ -18,7 +18,7 @@ enum {
// --- atom: floor_f3_face (58 words) --- // --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25 #define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 13 #define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum { enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit, atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
+29
View File
@@ -0,0 +1,29 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Source: C:\projects\Pikuma\ps1\code\hello_gte\hello_gte.tape.c
#pragma once
#pragma region hello_gte.tape
// --- atom: cube_g4_face (77 words) ---
#define _atom_offset_cull_cube_g4_face_exit 42
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
};
// --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
};
#pragma endregion hello_gte.tape
@@ -20,10 +20,10 @@
#include "duffle/lottes_tape.h" #include "duffle/lottes_tape.h"
#include "duffle/word_count.metadata.h" #include "duffle/word_count.metadata.h"
# include "gen/gte_hello.offsets.h" # include "gen/hello_gte.offsets.h"
#include "hello_gte.h" #include "hello_gte.h"
#include "hello_gte_tape.c" #include "hello_gte.tape.c"
typedef U4 OrderingTable_Buffer[OrderingTbl_Len]; typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2); typedef Array_(OrderingTable_Buffer, 2);
@@ -122,7 +122,7 @@ global SMemory smem;
extern SMemory smem; extern SMemory smem;
// TODO(Ed): // TODO(Ed):
FI_ U4* spad_warm(MipsAtom atom) { FI_ U4* spad_warm(Slice_MipsCode atom) {
return nullptr; return nullptr;
} }
+218
View File
@@ -0,0 +1,218 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "duffle/gen/duffle.macs.h"
# include "duffle/gen/duffle.offsets.h"
# include "duffle/atom_dsl.h"
# include "duffle/lottes_tape.h"
# include "duffle/word_count.metadata.h"
# include "gen/hello_gte.offsets.h"
# include "hello_gte.h"
#endif
#pragma region MACs (Mips Atom components)
#pragma endregion MACs
#pragma region Baked Atoms
/* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() };
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0),
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
store_word( R_AT, R_T7, 4),
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
add_ui( R_T1, R_0, 10),
shift_lleft_self(R_T1, S_(U4)/2),
add_u_self( R_T1, R_T6),
load_word( R_AT, R_T1, 0),
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
store_word( R_AT, R_T7, 0),
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
or_u_self( R_AT, R_V0),
store_word( R_AT, R_T1, 0),
add_ui(R_T7, R_T7, 20),
mac_yield()
};
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) {
/* Load 3 indices */
load_half_u(R_T0, R_T4, 0),
load_half_u(R_T1, R_T4, 2),
load_half_u(R_T2, R_T4, 4),
/* Load Vertices into GTE */
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
/* Run Math */
nop2, gte_cmdw_rtpt,
nop2, gte_cmdw_nclip,
nop2,
/* Advance Face Cursor and Yield */
add_ui(R_T4, R_T4, 8),
mac_yield()
};
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(),
gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3(),
gte_cmdw_avg_sort_z4,
gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(),
mac_format_g4_color(
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
// end: branch(bounds_chk)
// end: branch(cull)
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
// atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices( R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk)
// end: branch(culling)
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
#pragma endregion Baked Atoms
+71
View File
@@ -0,0 +1,71 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua — DO NOT EDIT
// Source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.tape.c
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
#ifndef WORD_COUNT
#define WORD_COUNT(name, count) enum { words_##name = (count) };
#endif
/* atom_dbg_skip */
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
load_half( rs_x, r_base, O_(V3_S2,x)) \
, load_half( rs_y, r_base, O_(V3_S2,y))
WORD_COUNT(mac_load_v2s2, 2)
/* atom_dbg_skip */
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
store_half(rt_x, base, offset + O_(V2_S2,x)) \
, store_half(rt_y, base, offset + O_(V2_S2,y))
WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
, store_half(rt_height, base, offset + O_(Rect_S2,height))
WORD_COUNT(mac_store_rects2, 4)
/* atom_dbg_skip */
#define mac_store_rgb8(rr, rg, rb, base, offset) \
store_byte(rr, base, offset + O_(DrawEnv,initial_bg_color.r)) \
, store_byte(rg, base, offset + O_(DrawEnv,initial_bg_color.g)) \
, store_byte(rb, base, offset + O_(DrawEnv,initial_bg_color.b))
WORD_COUNT(mac_store_rgb8, 3)
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
load_upper_i(reg_transfer, cmd >> 16) \
, or_i_self( reg_transfer, cmd & 0xFFFF) \
, store_word( reg_transfer, reg_base, port)
WORD_COUNT(mac_gcmd_push, 3)
#define mac_put_disp_env(reg_transfer, reg_base, port) \
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
WORD_COUNT(mac_put_disp_env, 15)
#define mac_put_draw_env(reg_transfer, reg_base, port) \
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
WORD_COUNT(mac_put_draw_env, 48)
@@ -0,0 +1,73 @@
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
// Source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.tape.c
#pragma once
#pragma region hello_joypad.tape
// --- atom: cube_g4_face (77 words) ---
#define _atom_offset_cull_cube_g4_face_exit 42
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
enum {
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
};
// --- atom: floor_f3_face (58 words) ---
#define _atom_offset_culling_floor_f3_face_exit 25
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
enum {
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
};
// --- atom: pad_bios_snapshot (78 words) ---
#define _atom_offset_snap_root_skip_disconnected 8
#define _atom_offset_disconnected_snap_end 60
#define _atom_offset_case_2_id_dispatch 8
#define _atom_offset_pending_snap_end 50
#define _atom_offset_id_dispatch_try_analog_stick 11
#define _atom_offset_id_dispatch_snap_end 37
#define _atom_offset_try_analog_stick_try_analog_pad 12
#define _atom_offset_analog_stick_snap_end 23
#define _atom_offset_try_analog_pad_try_unsupported 11
#define _atom_offset_analog_pad_snap_end 9
enum {
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
};
// --- atom: pad_apply_input (60 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6
#define _atom_offset_dead_zone_low_check_dead_low_active 8
#define _atom_offset_dead_zone_high_check_dead_high_active 15
#define _atom_offset_dead_zone_skip_exit_stick 23
#define _atom_offset_end_low_exit_stick 11
enum {
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
};
#pragma endregion hello_joypad.tape
+545
View File
@@ -0,0 +1,545 @@
#include <stdio.h>
#include <stdlib.h>
#include <assert.h>
// #include "libgpu.h"
// #include "libetc.h"
// #include "libgte.h"
#include "duffle/dsl.h"
#include "duffle/memory.h"
#include "duffle/math.h"
#include "duffle/gcc_asm.h"
#include "duffle/mips.h"
#include "duffle/gp.h"
#include "duffle/gte.h"
#include "duffle/pad.h"
# include "duffle/gen/duffle.macs.h"
# include "duffle/gen/duffle.offsets.h"
#include "duffle/atom_dsl.h"
#include "duffle/lottes_tape.h"
#include "duffle/word_count.metadata.h"
#include "psyq.h"
# include "gen/hello_joypad.macs.h"
# include "gen/hello_joypad.offsets.h"
#include "hello_joypad.h"
#include "psyq.c"
#include "hello_joypad.tape.c"
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2);
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
typedef Array_(PrimitiveBuffer, 2);
typedef Struct_(PrimitiveArena) {
A2_PrimitiveBuffer buf;
U4 used;
};
#define Cube_num_verts 8
typedef Array_(V3_S2, Cube_num_verts);
#define Cube_num_faces 6
typedef Array_(V4_S2, Cube_num_faces);
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
{ -128, -128, -128 },
{ 128, -128, -128 },
{ 128, -128, 128 },
{ -128, -128, 128 },
{ -128, 128, -128 },
{ 128, 128, -128 },
{ 128, 128, 128 },
{ -128, 128, 128 }
};
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
{ 3, 2, 0, 1 },
{ 0, 1, 4, 5 },
{ 4, 5, 7, 6 },
{ 1, 2, 5, 6 },
{ 2, 3, 6, 7 },
{ 3, 0, 7, 4 },
};
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
return;
}
typedef Struct_(Ent_Cube) {
V3_S4 accel;
V3_S4 vel;
V3_S4 pos;
V3_S4 scale;
V3_S2 rot;
A8_V3_S2 verts;
A6_V4_S2 faces;
};
#define Floor_num_verts 4
typedef Array_(V3_S2, Floor_num_verts);
#define Floor_num_faces 2
typedef Array_(V3_S2, Floor_num_faces);
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
{ -900, 0, -900 },
{ -900, 0, 900 },
{ 900, 0, -900 },
{ 900, 0, 900 },
};
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
{ 0, 1, 2 },
{ 1, 3, 2 },
};
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
};
typedef Struct_(Ent_Floor) {
V3_S4 accel;
V3_S4 pos;
V3_S4 scale;
V3_S2 rot;
A4_V3_S2 verts;
A2_V3_S2 faces;
};
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
};
typedef Struct_(SMemory) {
U4 MemTape[MemTape_Len];
DoubleBuffer screen_buf;
A2_OrderingTable_Buffer ordering_tbl;
PrimitiveArena primitives;
S4 active_buf_id;
M3_S2 tform_world;
Ent_Cube cube;
Ent_Floor floor;
PadBiosRaw pad_raw[2];
PadState pad[2];
U4_V scratchpad; // d-cache
};
global SMemory smem;
extern SMemory smem;
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
gknown PrimitiveArena* pa = & smem.primitives;
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
assert(pa->used + type_width < PrimitiveBuff_Len);
B1* next = buf + pa->used;
pa->used += type_width;
return next;
}
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
*
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
* The C-level writes after the call re-load the pointers from their callee-saved homes.
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
// Use enums.
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
* $a0 = raw0 (rgcc-bound; survives the sequence below)
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
u1_v(raw0)[0] = 0xFF;
u1_v(raw1)[0] = 0xFF;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
}
void gp_screen_init_c11(DoubleBuffer* screen_buf, S4* active_buf_id)
{
reset_graph(0);
// Set the current initial buffer
active_buf_id[0] = 0;
// Just setting env data, not interacting with console hw.
// First buffer area
displayenv_init(& r_(screen_buf->display)[0], 0, 0, ScreenRes_X, ScreenRes_Y);
drawenv_init (& r_(screen_buf->draw )[0], 0, ScreenRes_Y, ScreenRes_X, ScreenRes_Y);
// Second buffer area
displayenv_init(& r_(screen_buf->display)[1], 0, ScreenRes_Y, ScreenRes_X, ScreenRes_Y);
drawenv_init (& r_(screen_buf->draw )[1], 0, 0, ScreenRes_X, ScreenRes_Y);
// Set the back/drawing buffer
screen_buf->draw[0].enable_auto_clear = true;
screen_buf->draw[1].enable_auto_clear = true;
// Set the background clear color
screen_buf->draw[0].initial_bg_color = rgb8( .r = 7, .g = 7, .b = 7 );
screen_buf->draw[1].initial_bg_color = rgb8( .r = 7, .g = 7, .b = 7 );
// screen_buf->draw[1].initial_bg_color = rgb8( .r = 47, .g = 13, .b = 0 );
displayenv_put(& r_(screen_buf->display)[ active_buf_id[0] ]);
drawenv_put (& r_(screen_buf->draw )[ active_buf_id[0] ]);
// Initialize and setup the GTE geometry offsets
geom_init();
geom_set_offset(ScreenRes_CenterX, ScreenRes_CenterY);
geom_set_screen(ScreenZ);
set_display_enabled(1); // gp_DisplayEnabled
}
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
draw_sync(0);
vsync(0);
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
{
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
pa->used = 0;
}
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
}
GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf)
{
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
if (0) // Pad Input (dead — kept for the source-as-written record; references the deleted `pad_state` field)
{
(void)Pad_Left; (void)Pad_Right; /* suppress unused-token warnings */
if (false) {
smem.cube.rot.y += 30;
smem.floor.rot.y += 5;
}
if (false) {
smem.cube.rot.y -= 30;
smem.floor.rot.y -= 5;
}
}
if (1) // Pad Input (Tape version)
{
tb.used = 0; tb_scope_run(& tb) {
/* BIOS-owned polling: per-frame snapshot of both ports. */
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]);
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]);
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
tb_emit_(pad_apply_input);
tb_data_(state, & smem.pad[0]);
tb_data_(cube_rot, & smem.cube.rot);
tb_data_(floor_rot, & smem.floor.rot);
}
}
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
// Update the position based on acceleration and velocity
gknown V3_S4_R pos = & smem.cube.pos;
gknown V3_S4_R vel = & smem.cube.vel;
gknown V3_S4_R acc = & smem.cube.accel;
add_v3s4(vel, acc[0]);
add_v3s4_fp(pos, vel[0]);
// vel->x += acc->x;
// vel->y += acc->y;
// vel->z += acc->z;
// pos->x += vel->x;
// pos->y += vel->y;
// pos->z += vel->z;
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
// Prep
S4 nclip = 0;
S4 orderingtbl_z = 0;
A2_S2 p; //???
S4 flag; //????
// Draw Cube
if (0)
{
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.cube.pos);
m3s2_scale (& smem.tform_world, & smem.cube.scale);
// gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
{
Poly_G4* quad = prim_alloc(Poly_G4); set_poly_g4(quad);
quad->c0 = rgb8(255, 0, 255);
quad->c1 = rgb8(255, 255, 0);
quad->c2 = rgb8( 0, 255, 255);
quad->c3 = rgb8( 0, 255, 0);
V4_S2* face = & smem.cube.faces[face_id];
V3_S2* p0 = & smem.cube.verts[face->x];
V3_S2* p1 = & smem.cube.verts[face->y];
V3_S2* p2 = & smem.cube.verts[face->z];
V3_S2* p3 = & smem.cube.verts[face->w];
nclip = rtp_avg_nclip_a4_v3s2(
p0, p1, p2, p3,
& quad->p0, & quad->p1, & quad->p2, & quad->p3,
& p, & orderingtbl_z, & flag
);
if (nclip <= 0) {
continue;
}
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], quad);
}
}
// smem.cube.rot.x += 6;
// smem.cube.rot.y += 8;
// smem.cube.rot.z += 12;
smem.cube.rot.y += 30;
}
// Draw cube (tape method) - two triangles per face
if (1)
{
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.cube.pos);
m3s2_scale (& smem.tform_world, & smem.cube.scale);
gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
tb.used = 0; tb_scope(& tb) {
tb_emit(& tb, rbind_cube_g4_face);
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.cube.faces));
tb_data(& tb, u4_(smem.cube.verts));
tb_data(& tb, u4_(ordering_buf));
for (U4 i = 0; i < Cube_num_faces; i++) {
// Two triangles per quad face: (x,y,z) and (x,z,w)
tb_emit(& tb, cube_g4_face);
}
tb_emit(& tb, sync_primitive_arena);
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run(tb_slice(tb));
// smem.cube.rot.y += 30;
}
// Draw Floor
if (0)
{
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.floor.pos);
m3s2_scale (& smem.tform_world, & smem.floor.scale);
gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world);
for (U4 face_id = 0; face_id < Floor_num_faces; face_id += 1)
{
Poly_F3* tri = prim_alloc(Poly_F3); set_poly_f3(tri);
tri->color = rgb8(255, 255, 255);
V3_S2* face = & smem.floor.faces[face_id];
register V3_S2* p0 rgcc(R_T4) = & smem.floor.verts[face->x];
register V3_S2* p1 rgcc(R_T5) = & smem.floor.verts[face->y];
register V3_S2* p2 rgcc(R_T6) = & smem.floor.verts[face->z];
gte_load_v0(p0, R_T4);
/*
asm volatile( ".word " "%0" ", %1" : :
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_xy & REG_MASK) << RT_SHIFT) | (0 & IMM_MASK)),
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_z & REG_MASK) << RT_SHIFT) | (GTE_Z_Offset & IMM_MASK)),
"r"(p0) :
"$2", "$8", "$9", "$31", "memory"
);
*/
gte_load_v1(p1, R_T5);
gte_load_v2(p2, R_T6);
gte_rtpt();
gte_nclip();
gte_stotz(& nclip);
// nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
// , & tri->p0, & tri->p1, & tri->p2
// , & p, & orderingtbl_z, & flag
// );
// if (nclip <= 0) {
// continue;
// }
if (nclip > 0 ) {
gte_stsxy3(& tri->p0, & tri->p1, & tri->p2);
gte_avsz3();
gte_stotz(& orderingtbl_z);
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
}
}
}
smem.floor.rot.y += 5;
}
// Draw floor tape method
if (1)
{
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.floor.pos);
m3s2_scale (& smem.tform_world, & smem.floor.scale);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used;
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
// The tape atoms in-flight should not need to care.
// Prepare the tape. (Push protocol to tape)
tb.used = 0; tb_scope(& tb) {
tb_emit(& tb, set_gte_world);
tb_data(& tb, u4_(& smem.tform_world));
tb_emit(& tb, rbind_floor_f3_face);
// TODO(Ed): Just use a single context struct ref
tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.floor.faces));
tb_data(& tb, u4_(smem.floor.verts));
tb_data(& tb, u4_(ordering_buf));
for (U4 i = 0; i < Floor_num_faces; i++) {
tb_emit(& tb, floor_f3_face);
}
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
tb_emit(& tb, sync_primitive_arena);
tb_data(& tb, u4_(& pa->used));
tb_data(& tb, prim_base);
}
tape_run(tb_slice(tb));// Fire off the tape.
// C-side state (pa->used) has already been updated by the tape!
// smem.floor.rot.y += 5;
}
// --- TAPE DIAGNOSTICS ---
if (0)
{
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
// Skip set_gte_world atom for diagnostics to isolate the triangle loop
for (U4 i = 0; i < Floor_num_faces; i++) {
// tb_emit(& tb, code_diag_yield);
// tb_emit(& tb, code_diag_color);
// tb_emit(& tb, code_diag_gte);
}
}
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
tape_run(tb_slice(tb));
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
}
}
GCC_OPTIMIZATION_ENABLE
void render(void) {
}
int main(void)
{
smem = (SMemory){0};
smem.scratchpad = C_(U4_V, 0x1F800000);
// smem.primitives.used = 0;
// smem.active_buf_id = 0;
/*Persistent Entity Setup*/{
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
Ent_Cube* cube = & smem.cube;
cube->rot = v3s2(0, 0, 0);
cube->scale = v3s4_fp_one();
cube->accel = v3s4(0, 1, 0);
cube->pos = v3s4(0, -400, 1800);
}
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
Ent_Floor* floor = & smem.floor;
floor->rot = v3s2(0, 0, 0);
floor->pos = v3s4(0, 450, 1800);
floor->scale = v3s4_fp_one();
}
}
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
reset_graph(0);
/* Direct BIOS: poll both ports during VBlank. */
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
/* Pinned registers for the GPU init atom. */
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
tb.used = 0; tb_scope_run(& tb) {
tb_emit(& tb, screen_env_init);
tb_emit(& tb, gp_screen_init);
}
}
while (1) {
gknown S4* active_buf_id = & smem.active_buf_id;
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
gknown PrimitiveArena* pa = & smem.primitives;
update(pa, ordering_buf);
render();
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
};
return 0;
}
+26
View File
@@ -0,0 +1,26 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "duffle/dsl.h"
# include "duffle/math.h"
# include "duffle/gp.h"
# include "duffle/pad.h"
#endif
enum {
PrimitiveBuff_Len = 4096,
OrderingTbl_Len = 2048
};
enum {
ScreenRes_X = 320,
ScreenRes_Y = 240,
ScreenZ = 320,
ScreenRes_CenterX = (ScreenRes_X >> 1),
ScreenRes_CenterY = (ScreenRes_Y >> 1),
};
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
+705
View File
@@ -0,0 +1,705 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "duffle/gen/duffle.macs.h"
# include "duffle/gen/duffle.offsets.h"
# include "duffle/atom_dsl.h"
# include "duffle/lottes_tape.h"
# include "duffle/mips.h"
# include "duffle/gte.h"
# include "duffle/gp.h"
# include "duffle/pad.h"
# include "duffle/word_count.metadata.h"
# include "psyq.h"
# include "gen/hello_joypad.offsets.h"
# include "gen/hello_joypad.macs.h"
# include "hello_joypad.h"
#endif
#pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_load_v2s2(U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, {
load_half( rs_x, r_base, O_(V3_S2,x)),
load_half( rs_y, r_base, O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
store_half(rt_height, base, offset + O_(Rect_S2,height)),
})
FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, {
store_byte(rr, base, offset + O_(DrawEnv,initial_bg_color.r)),
store_byte(rg, base, offset + O_(DrawEnv,initial_bg_color.g)),
store_byte(rb, base, offset + O_(DrawEnv,initial_bg_color.b)),
})
FI_ Slice_MipsCode ac_gcmd_push(U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_gcmd_push, {
load_upper_i(reg_transfer, cmd >> 16),
or_i_self( reg_transfer, cmd & 0xFFFF),
store_word( reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, {
/*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References:
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
*
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
* ./toolchain/psyq-4_7/lib/libgpu.a
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
*
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
* code[8..10] = padding (NOP) — 3 words to fill the packet
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
* code[13..14] = padding (NOP) — completes the 16-word packet
*/
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
})
#pragma endregion MACs
#pragma region Baked Atoms
/* DIAGNOSTIC 1: Pure tape loop test */
internal MipsAtom_(diag_yield) { mac_yield() };
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
internal MipsAtom_(diag_color) {
store_word( R_0, R_T7, 0),
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
store_word( R_AT, R_T7, 4),
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
add_ui( R_T1, R_0, 10),
shift_lleft_self(R_T1, S_(U4)/2),
add_u_self( R_T1, R_T6),
load_word( R_AT, R_T1, 0),
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
store_word( R_AT, R_T7, 0),
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
or_u_self( R_AT, R_V0),
store_word( R_AT, R_T1, 0),
add_ui(R_T7, R_T7, 20),
mac_yield()
};
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
internal MipsAtom_(diag_gte) {
/* Load 3 indices */
load_half_u(R_T0, R_T4, 0),
load_half_u(R_T1, R_T4, 2),
load_half_u(R_T2, R_T4, 4),
/* Load Vertices into GTE */
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
/* Run Math */
nop2, gte_cmdw_rtpt,
nop2, gte_cmdw_nclip,
nop2,
/* Advance Face Cursor and Yield */
add_ui(R_T4, R_T4, 8),
mac_yield()
};
enum {
R_ScreenX = R_T5 atom_reg atom_type(U2),
R_ScreenY = R_T6 atom_reg atom_type(U2),
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
#define R_ScreenBuf_Code R_T7_Code
};
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
) {
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
add_ui(R_T0, R_0, gp0_tpage_default),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
add_ui(R_T0, R_0, 1),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
add_ui(R_T0, R_0, 7),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
mac_yield(),
};
enum {
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
#define R_IO_BaseAddr_Code R_T4_Code
};
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU→GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
/* GP1: DisplayMode + Display Ranges */
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
mac_yield(),
};
typedef Struct_(Binds_CubeTri) {
U4 PrimCursor;
V4_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
atom_writes(R_PrimCursor, R_FaceCursor)
){
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(),
gte_cmdw_rotate_translate_perspective_single,
mac_gte_store_g4_p3(),
gte_cmdw_avg_sort_z4,
gte_mv_from_data_r(R_T1, C2_OTZ),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(),
mac_format_g4_color(
/* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00,
/* c2 cyan */ 0x00, 0xFF, 0xFF,
/* c3 green */ 0x00, 0xFF, 0x00),
// end: branch(bounds_chk)
// end: branch(cull)
atom_label(cube_g4_face_exit)
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
mac_yield()
};
typedef Struct_(Binds_FloorTri) {
U4 PrimCursor;
V3_S2* FaceCursor;
V3_S2* VertBase;
U4* OtBase;
};
internal
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
, atom_reads(R_TapePtr)
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
){
/* Pop 4 arguments from the tape directly into the workspace registers */
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
// atom_dbg_skip
internal
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
, atom_writes(R_PrimCursor, R_FaceCursor)
) {
mac_load_tri_indices( R_T0, R_T1, R_T2),
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
gte_cmdw_nclip,
/* Culling (Branch forward if Backface) */
gte_mv_from_data_r(R_T0, C2_MAC0),
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
/* Format Primitive */
mac_gte_store_f3(),
/* Calculate Depth */
gte_avg_sort_z3,
gte_mv_from_data_r(R_T1, C2_OTZ),
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk)
// end: branch(culling)
/* Advance Input Cursor & Yield (Both branch targets land here) */
atom_label(floor_f3_face_exit)
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
mac_yield()
};
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
mac_yield()
};
/* ----- pad_bios_snapshot -----
* Per-frame snapshot of one BIOS pad buffer into PadState.
* Decoder (branch ladder on raw[0] status + raw[1] id):
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
* 6. else -> Unsupported (buttons=0, axes=0x80)
*
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
*
* Register use (atom-local; no wave-context touched):
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
* R_T1 = state base (kept throughout; all stores go through R_T1)
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
* R_T4 = scratch (shifts, compares, immediate loads, store values)
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
*/
enum {
R_PadRaw = R_T0 atom_reg atom_type(U1),
R_PadState = R_T1 atom_reg,
R_RawStatus = R_T2 atom_reg,
R_RawId = R_T3 atom_reg,
};
typedef Struct_(Binds_PadBiosSnapshot) {
PadBiosRaw* raw;
PadState* state;
};
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
) {
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
/* === Read raw[0] (status) + raw[1] (id) */
load_byte_u(R_RawStatus, R_PadRaw, 0),
load_byte_u(R_RawId, R_PadRaw, 1),
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
atom_label(disconnected) /* === Disconnected body. */
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
branch_equal(R_0, R_0, atom_offset(disconnected, snap_end)), nop,
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(disconnected, snap_end)), nop,
atom_label(skip_disconnected)
/* === Case 2: Pending (status == 0 && id == 0)
* Combined check: if (status | id) != 0 then skip to id_dispatch.
* Falls through to the Pending case only when both are zero. */
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
atom_label(pending) /* === Pending body */
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
branch_equal(R_0, R_0, atom_offset(pending, snap_end)), nop,
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(pending, snap_end)), nop,
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
store_word( R_T5, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0x41),
store_byte( R_T4, R_PadState, O_(PadState,id)),
branch_equal(R_0, R_0, atom_offset(id_dispatch, snap_end)), nop,
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(id_dispatch, snap_end)), nop,
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
atom_label(analog_stick) /* === AnalogStick body
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
store_half( R_T4, R_PadState, O_(PadState,right_x)),
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
store_byte( R_T5, R_PadState, O_(PadState,id)),
branch_equal(R_0, R_0, atom_offset(analog_stick, snap_end)), nop,
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(analog_stick, snap_end)), nop,
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
and_i( R_T4, R_RawId, 0xF0),
add_ui( R_T5, R_0, 0x70),
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
atom_label(analog_pad) /* === AnalogPad body
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
store_word( R_T4, R_PadState, O_(PadState,status)),
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
store_half( R_T4, R_PadState, O_(PadState,buttons)),
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
store_half( R_T4, R_PadState, O_(PadState,right_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
branch_equal(R_0, R_0, atom_offset(analog_pad, snap_end)), nop,
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(analog_pad, snap_end)), nop,
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
add_ui( R_T4, R_0, PadStatus_Unsupported),
store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
store_byte( R_T4, R_PadState, O_(PadState,id)),
/* Fall through to snap_end. */
atom_label(snap_end)
mac_yield(),
};
/* ----- pad_apply_input -----
* Reads pad[0].buttons + pad[0].left_x;
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
* - Analog stick X (dead zone 0x70..0x90):
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
* - D-pad + analog deltas add when used together.
*
* Convention:
* pad_state = 0 means no buttons active.
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
*
* Signed-delta trick:
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
*/
typedef Struct_(Binds_PadApplyInput) {
PadState* state;
V3_S2* cube_rot;
V3_S2* floor_rot;
};
enum {
R_PadStateT5 = R_T5 atom_reg,
R_CubeRot = R_T1 atom_reg,
R_FloorRot = R_T2 atom_reg,
};
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
, atom_writes( R_CubeRot, R_FloorRot)
) {
/* Pop Binds from tape (state, cube_rot, floor_rot) */
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
/* Load pad[0].buttons into R_T0. */
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
// Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30),
add_si( R_T3, R_T3, 5),
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30),
add_si( R_T3, R_T3, -5),
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_dpad_right)
/* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
add_ui( R_T4, R_0, 0x90),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
branch_equal(R_0, R_0, atom_offset(dead_zone_skip, exit_stick)), nop,
/* Fall-through = left_x in [0x70, 0x90] (dead zone); skip analog entirely. */
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(dead_zone_skip, exit_stick)), nop,
atom_label(dead_low_active)
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
/* delta = 0x80 - left_x (positive). */
/* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
branch_equal(R_0, R_0, atom_offset(end_low, exit_stick)), nop,
// TODO(Ed): Lua metaprogram: Support jump instruction here..
// jump(atom_offset(end_low, exit_stick)), nop,
atom_label(dead_high_active)
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
sub_u( R_T3, R_T4, R_T3),
/* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
atom_label(exit_stick)
mac_yield(),
};
#pragma endregion Baked Atoms
+3
View File
@@ -0,0 +1,3 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "psyq.h"
#endif
+103
View File
@@ -0,0 +1,103 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
# include "duffle/dsl.h"
# include "duffle/math.h"
# include "duffle/gp.h"
#endif
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
typedef Struct_(DrawEnv) {
Rect_S2 clip_area;
V2_S2 drawing_offset[2];
Rect_S2 texture_window;
S2 texture_page;
B1 flag_dither;
B1 flag_draw_on_display;
B1 enable_auto_clear;
RGB8 initial_bg_color;
DrawEnv_Packed dr_env; // reserved
};
typedef Struct_(DisplayEnv) {
Rect_S2 display_area;
Rect_S2 screen;
B1 vinterlace;
B1 color24;
B1 pad0;
B1 pad1;
};
typedef Array_(DrawEnv, 2);
typedef Array_(DisplayEnv, 2);
typedef Struct_(DoubleBuffer) {
A2_DrawEnv draw;
A2_DisplayEnv display;
};
DisplayEnv* displayenv_init(DisplayEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDispEnv");
DrawEnv* drawenv_init (DrawEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDrawEnv");
DisplayEnv* displayenv_put(DisplayEnv* env) asm("PutDispEnv");
DrawEnv* drawenv_put (DrawEnv* env) asm("PutDrawEnv");
U4 geom_init(void) asm("InitGeom");
void geom_set_offset(U4 x, U4 y) asm("SetGeomOffset");
void geom_set_screen(U4 h) asm("SetGeomScreen");
U4* orderingtbl_clear_reverse(U4* ot, U4 len) asm("ClearOTagR");
U4 reset_graph(U4 mode) asm("ResetGraph");
void set_display_enabled(U4 mask) asm("SetDispMask");
U4 draw_sync(U4 mode) asm("DrawSync");
U4 vsync(U4 mode) asm("VSync");
void draw_orderingtbl(U4* buf) asm("DrawOTag");
typedef Struct_(Tile) {
U4 tag;
RGB8 color;
B1 code;
Rect_S2 rect;
};
/*
Linear Algebra
*/
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix");
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix");
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix");
// Rotation, Translation, Perspective
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) asm("RotTransPers");
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip3");
FI_ S4 rtp_avg_nclip_a3_v3s2(
V3_S2* v0, V3_S2* v1, V3_S2* v2,
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2,
A2_S2* pp, S4* otz, S4* flag
){
return rtp_avg_nclip_a3_v3s2_raw(
v0, v1, v2,
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
);
}
S4 rtp_avg_nclip_a4_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3, S4* xy1, S4* xy2, S4* xy3, S4* xy4, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip4");
FI_ S4 rtp_avg_nclip_a4_v3s2(
V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3,
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2, V2_S2* xy3,
A2_S2* pp, S4* otz, S4* flag
){
return rtp_avg_nclip_a4_v3s2_raw(
v0, v1, v2, v3,
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
);
}
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix");
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix");
+659
View File
@@ -0,0 +1,659 @@
#if 0 /* ac_pad_sio_write_pad_state — superseded by pad_bios_snapshot */
/* ============================================================
* raw_sio_pad_poll_20260802 — superseded by bios_pad_buffer_snapshot_20260803.
* The doomed raw-SIO production atoms (ac_pad_sio_write_pad_state,
* pad_sio_init, pad_sio_step, pad_sio_diag_pin, pad_sio_diag_byte_exchange)
* reference symbols that were removed from code/duffle/pad.h during
* Phase 1. Each is wrapped in a narrow `#if 0` so the C compile skips
* the body while the source-as-written text stays in place for the
* Phase 5.1 deletion pass. The wrap is removed (and the bodies are
* deleted) by Phase 5.1 of this track.
* ============================================================ */
* Writes the per-port PadState in 5 instructions plus 4 store_word calls (status,
* buttons, left_x/y/right_x/right_y packed, attempt). The provisional decode publishes
* 0x0000FFFF buttons + centered axes on every path until response-byte decode lands.
*
* Args:
* status_val - the PadSioStatus enum value to publish
* state_ptr_reg - the PadState* base (R_PadState at the call site)
* scratch_reg - scratch register for the value being stored (e.g., R_T0)
*
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
* two-instruction zero-extended buttons load).
*/
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, {
add_ui(scratch_reg, R_0, status_val),
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
* libetc convention. Build it with LUI + ORI so addiu does not
* sign-extend 0xFFFF to 0xFFFFFFFF. */
load_upper_i(scratch_reg, 0x0000),
or_i(scratch_reg, scratch_reg, 0xFFFF),
store_word(scratch_reg, state_ptr_reg, O_(PadState,buttons)),
add_ui(scratch_reg, R_0, 0x80808080),
store_word(scratch_reg, state_ptr_reg, O_(PadState,left_x)),
add_ui(scratch_reg, R_0, 0),
store_word(scratch_reg, state_ptr_reg, O_(PadState,attempt))
})
#endif /* end ac_pad_sio_write_pad_state wrap */
/* ----- pad_sio_init -----
* Boot-time SIO0 init. Caller pins R_T6 = sio_base_addr0.
* Issues SIO CTRL=0x0040 (reset), MODE=0x000D, BAUD=0x0088.
* (Phase 2 fills the body.)
*/
#if 0 /* pad_sio_init — superseded by pad_bios_init_start (Phase 1.3) */
internal MipsAtom_(pad_sio_init) atom_info(atom_phase(pad_init)
, atom_reads(R_T5, R_T6)
, atom_writes(R_T5, R_T6)
) {
/* FIX 2026-08-02: explicitly load the KSEG1 base into R_T6 at the top of
* the atom body. The rgcc(R_PadSioBase) binding in main() pins R_T6 = base
* when main() runs, but $12 is caller-saved per the O32 ABI — when tape_run
* is invoked, R_T6 is fair game. The atom body cannot rely on the value. */
load_upper_i(R_T6, pad_IO_KSEG1_BASE >> 16), /* R_T6 high 16 = 0xBF80 */
or_i(R_T6, R_T6, pad_IO_KSEG1_BASE & 0xFFFF), /* R_T6 = 0xBF800000 */
/* SIO CTRL = 0x0040 (reset) */
add_ui(R_T5, R_0, pad_SIO_CTRL_RESET),
store_half(R_T5, R_T6, pad_SIO_CTRL_OFFSET),
/* SIO MODE = 0x000D (MUL1, 8-bit, no parity, idle-high) */
add_ui(R_T5, R_0, pad_SIO_MODE_INIT),
store_half(R_T5, R_T6, pad_SIO_MODE_OFFSET),
/* SIO BAUD = 0x0088 (~250 kHz) */
add_ui(R_T5, R_0, pad_SIO_BAUD_INIT),
store_half(R_T5, R_T6, pad_SIO_BAUD_OFFSET),
mac_yield(),
};
#endif /* end pad_sio_init wrap */
/* ----- pad_sio_step -----
* Per-frame bounded raw-SIO transaction. Reads PadState pointers + SIO
* base addresses from Binds_PadSioStep; writes per-port status +
* buttons + axes into smem.pad[0..1].
* Body shape (per spec §"Transaction model (per port, per pad_sio_step)"):
* port 0: CTRL=CLEANUP → settle → CTRL=port-select → settle → exchange 5
* bytes (addr + 0x42 0x00 0x00 0x00) → decode → write PadState[0]
* → CTRL=CLEANUP.
* port 1: swap scratch regs (sio_base_addr1 → R_PadSioBase, state1 →
* R_PadState) → mirror port 0 sequence.
*
* Bounded-loop semantics: every countdown is wrapped in
* add_ui_self(R_T1, -1) + branch_ne(R_T1, R_0, ...)
* with a known maximum (pad_SIO_SETTLE_BEFORE_TX=1000, pad_SIO_SETTLE_AFTER_TX=2000,
* pad_SIO_WAIT_BUDGET=4096). The static-analysis pass currently reports
* has_loops = true; the follow-up metaprogram track that learns modeled-bounded
* loops is out of scope here (per spec §"Risks").
*
* Scratch register strategy:
* R_PadStatus = R_T4 — RESERVED for port-1 swap (holds state1)
* R_PadCountdown = R_T5 — RESERVED for port-1 swap (holds sio_base_addr1)
* R_T0 — byte-exchange value + STAT read (clobbered freely)
* R_T1 — countdown budget (clobbered freely)
* R_PadState = R_T7 — PadState* (preserved for PadState writes)
* R_PadSioBase = R_T6 — SIO base (preserved through the port)
*
* Response decode (Task 3.1 teaching scope):
* - status = PadSioStatus_Digital (hardcoded)
* - buttons = 0xFFFF (no buttons pressed in the provisional libetc
* convention; full response-byte decode is follow-up)
* - axes = 0x80808080 (centered: left_x=0x80, left_y=0x80,
* right_x=0x80, right_y=0x80)
* - attempt = 0
* - DualShock handshake (0x43 0x01 → 0x44 0x01 0x03 → 0x43 0x00) is
* follow-up scope; the hardcoded digital decode is a placeholder.
*
* Both ports raise /CS (CTRL = pad_SIO_CTRL_CLEANUP) before exit. Both ports
* treat response timeout as PadSioStatus_Disconnected per the spec §"Failure
* handling" + the canonical per-port timeout semantics.
*/
#if 0 /* pad_sio_step — superseded by pad_bios_snapshot (Phase 2.1) */
internal MipsAtom_(pad_sio_step) atom_info(atom_bind(Binds_PadSioStep)
, atom_reads(R_TapePtr, R_PadSioBase, R_PadState, R_PadStatus, R_PadCountdown)
, atom_writes(R_PadStatus, R_PadCountdown)
) {
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
* top. The rgcc() binding in main() does NOT survive the tape_run call
* because R_T6 is caller-saved per the O32 ABI. The pad_sio_init atom
* (also in the per-frame tape) reloads R_T6 separately. */
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
/* Pop Binds from tape (in Binds_PadSioStep declaration order) */
load_word(R_PadState, R_TapePtr, O_(Binds_PadSioStep,state0)),
load_word(R_PadStatus, R_TapePtr, O_(Binds_PadSioStep,state1)), /* reserved for port-1 swap */
load_word(R_PadSioBase, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr0)),
load_word(R_PadCountdown, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr1)), /* reserved for port-1 swap */
add_ui_self(R_TapePtr, S_(Binds_PadSioStep)),
/* ============== PORT 0 TRANSACTION ============== */
/* Use R_T0 (byte value / STAT read) + R_T1 (countdown) as scratch.
* R_PadStatus (state1) + R_PadCountdown (sio_base_addr1) are preserved
* through the port-0 body and swapped into R_PadSioBase + R_PadState
* at atom_offset(port1_start, ...) below. */
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
atom_label(settle_pre_port0)
nop, /* BD slot */
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_pre_port0, settle_pre_port0)),
/* 2. Port-select: CTRL = 0x0003 (TX enable + DTR /CS) for port 0 */
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS), /* set /CS line low */
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
atom_label(settle_post_port0)
nop,
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_post_port0, settle_post_port0)),
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
add_ui(R_T0, R_0, pad_PROTO_ADDR),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack0_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port0, ack0_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack0_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port0, wait_ack0_port0)),
/* RX timeout → mark disconnected; skip to port 1 */
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack0)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack0, port1_start)),
atom_label(ack0_received_port0)
/* Read open-bus response byte 0 — discard per docs/psx-spx §controllersandmemorycards.md */
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Confirm RX FIFO drained before sending byte 1. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel0_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port0, ack_released_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel0_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port0, wait_ackrel0_port0)),
/* RX-drain timeout → disconnected; skip to port 1 */
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel0)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel0, port1_start)),
atom_label(ack_released_port0)
/* === Byte 1 (port 0): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack1_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port0, ack1_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack1_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port0, wait_ack1_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack1)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack1, port1_start)),
atom_label(ack1_received_port0)
/* Read response ID byte — discarded for teaching scope (decode hardcoded). */
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* RX FIFO drain wait. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel1_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port0, ack_released1_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel1_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port0, wait_ackrel1_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel1)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel1, port1_start)),
atom_label(ack_released1_port0)
/* === Byte 2 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack2_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port0, ack2_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack2_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port0, wait_ack2_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack2)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack2, port1_start)),
atom_label(ack2_received_port0)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel2_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port0, ack_released2_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel2_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port0, wait_ackrel2_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel2)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel2, port1_start)),
atom_label(ack_released2_port0)
/* === Byte 3 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack3_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port0, ack3_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack3_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port0, wait_ack3_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ack3)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack3, port1_start)),
atom_label(ack3_received_port0)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel3_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port0, ack_released3_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel3_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port0, wait_ackrel3_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_ackrel3)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel3, port1_start)),
atom_label(ack_released3_port0)
/* === Byte 4 (FINAL, port 0): send 0x00 + RX-not-empty wait + read final byte === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_rx4_port0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port0, rx4_received_port0)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_rx4_port0)
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port0, wait_rx4_port0)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port0_from_rx4)
branch_equal(R_0, R_0, atom_offset(skip_port0_from_rx4, port1_start)),
atom_label(rx4_received_port0)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
/* === RESPONSE DECODE (hardcoded for teaching scope) ===
* Per the plan §"Phase 3 task 3.1" + spec §"Architecture":
* - Full decode (buttons/axes from response bytes) is follow-up scope.
* - Teaching scope: hardcode digital poll response.
* status = PadSioStatus_Digital
* buttons = 0x0000FFFF (no buttons pressed — placeholder)
* axes = 0x80808080 (left_x=0x80, left_y=0x80, right_x=0x80, right_y=0x80)
* attempt = 0
*/
atom_label(decode_port0)
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
/* /CS cleanup: raise /CS, clear stale status before exiting port 0. */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* ============== PORT 1 SETUP ============== */
/* Swap: R_PadCountdown holds sio_base_addr1; R_PadStatus holds state1. */
atom_label(port1_start)
add_u(R_PadSioBase, R_0, R_PadCountdown), /* sio_base_addr1 → R_PadSioBase */
add_u(R_PadState, R_0, R_PadStatus), /* state1 → R_PadState */
/* ============== PORT 1 TRANSACTION (mirror of port 0) ============== */
/* R_PadStatus + R_PadCountdown are no longer reserved (port 1 is the
* last transaction); we still use R_T0/R_T1 as scratch to match port 0. */
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
atom_label(settle_pre_port1)
nop,
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_pre_port1, settle_pre_port1)),
/* 2. Port-select: CTRL = 0x0003 | (1 << 13) (port 1 select) */
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
or_i(R_T0, R_T0, 1 << 13), /* port 1 select bit (CTRL bit 13 = port select) */
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
atom_label(settle_post_port1)
nop,
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(settle_post_port1, settle_post_port1)),
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
add_ui(R_T0, R_0, pad_PROTO_ADDR),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack0_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port1, ack0_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack0_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port1, wait_ack0_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack0)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack0, end_atom)),
atom_label(ack0_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel0_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port1, ack_released_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel0_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port1, wait_ackrel0_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel0)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel0, end_atom)),
atom_label(ack_released_port1)
/* === Byte 1 (port 1): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack1_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port1, ack1_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack1_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port1, wait_ack1_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack1)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack1, end_atom)),
atom_label(ack1_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel1_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port1, ack_released1_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel1_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port1, wait_ackrel1_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel1)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel1, end_atom)),
atom_label(ack_released1_port1)
/* === Byte 2 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack2_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port1, ack2_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack2_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port1, wait_ack2_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack2)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack2, end_atom)),
atom_label(ack2_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel2_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port1, ack_released2_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel2_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port1, wait_ackrel2_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel2)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel2, end_atom)),
atom_label(ack_released2_port1)
/* === Byte 3 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ack3_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port1, ack3_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ack3_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port1, wait_ack3_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ack3)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack3, end_atom)),
atom_label(ack3_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_ackrel3_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port1, ack_released3_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_ackrel3_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port1, wait_ackrel3_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_ackrel3)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel3, end_atom)),
atom_label(ack_released3_port1)
/* === Byte 4 (FINAL, port 1): send 0x00 + RX-not-empty wait + read final byte === */
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
add_ui(R_T0, R_0, 0x00),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(wait_rx4_port1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port1, rx4_received_port1)),
add_ui_self(R_T1, -1),
atom_label(continue_wait_rx4_port1)
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port1, wait_rx4_port1)),
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
atom_label(skip_port1_from_rx4)
branch_equal(R_0, R_0, atom_offset(skip_port1_from_rx4, end_atom)),
atom_label(rx4_received_port1)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
/* === RESPONSE DECODE (port 1) === */
atom_label(decode_port1)
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
/* /CS cleanup: raise /CS, clear stale status before exiting port 1. */
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
atom_label(end_atom)
mac_yield(),
};
#endif /* end pad_sio_step wrap */
/* ----- pad_sio_diag_pin -----
* Per-frame diagnostic counter. The caller binds R_DiagPinScratch to
* scratch_for_atom_diag_pin for temporary gdb verification.
*/
#if 0 /* pad_sio_diag_pin — superseded (raw-SIO phase removed) */
internal MipsAtom_(pad_sio_diag_pin) atom_info(atom_phase(pad_init)
, atom_reads(R_T0, R_T1, R_DiagPinScratch)
, atom_writes(R_T0, R_T1, R_DiagPinScratch)
) {
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
load_upper_i(R_DiagPinScratch, 0x8001),
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
/* High half = 0xD1A6; low half increments once per atom invocation. */
load_word(R_T1, R_DiagPinScratch, 0),
nop,
add_ui(R_T1, R_T1, 1),
and_i(R_T0, R_T1, 0xFFFF),
load_upper_i(R_T1, 0xD1A6),
or_i(R_T1, R_T1, 0),
or_u(R_T1, R_T1, R_T0),
store_word(R_T1, R_DiagPinScratch, 0),
mac_yield(),
};
#endif /* end pad_sio_diag_pin wrap */
/* ----- pad_sio_diag_byte_exchange -----
* Temporary two-byte wire probe: sends 0x01 and 0x42, then stores the
* open-bus byte and response ID in scratch_for_atom_diag_pin.
*/
#if 0 /* pad_sio_diag_byte_exchange — superseded (raw-SIO phase removed) */
internal MipsAtom_(pad_sio_diag_byte_exchange) atom_info(atom_phase(pad_init)
, atom_reads(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
, atom_writes(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
) {
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
load_upper_i(R_DiagPinScratch, 0x8001),
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
* top. The rgcc() binding in main() does NOT survive the tape_run call
* because R_T6 is caller-saved per the O32 ABI. */
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
add_ui(R_T0, R_0, pad_PROTO_ADDR),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(diag_wait_ack0)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack0, diag_ack0_done)),
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack0, diag_wait_ack0)),
add_ui(R_T0, R_0, 0xDEADAC01),
store_word(R_T0, R_DiagPinScratch, 0),
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack0, diag_timeout)),
atom_label(diag_ack0_done)
load_byte_u(R_T2, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
atom_label(diag_wait_ack1)
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
nop,
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack1, diag_ack1_done)),
add_ui_self(R_T1, -1),
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack1, diag_wait_ack1)),
add_ui(R_T0, R_0, 0xDEADAC02),
store_word(R_T0, R_DiagPinScratch, 0),
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack1, diag_timeout)),
atom_label(diag_ack1_done)
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
nop,
shift_lleft(R_T0, R_T0, 8),
or_u(R_T2, R_T2, R_T0),
store_word(R_T2, R_DiagPinScratch, 0),
atom_label(diag_success)
branch_equal(R_0, R_0, atom_offset(diag_success, diag_done)),
nop,
atom_label(diag_timeout_ack0)
add_ui(R_T0, R_0, 0xDEADAC01),
store_word(R_T0, R_DiagPinScratch, 0),
atom_label(diag_timeout_ack1)
add_ui(R_T0, R_0, 0xDEADAC02),
store_word(R_T0, R_DiagPinScratch, 0),
atom_label(diag_timeout)
add_ui(R_T0, R_0, 0xDEADACFF),
store_word(R_T0, R_DiagPinScratch, 0),
atom_label(diag_done)
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
mac_yield(),
};
#endif /* end pad_sio_diag_byte_exchange wrap */
+7 -1
View File
@@ -7,12 +7,12 @@ A rest from the usual.
## Dependencies ## Dependencies
I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...): I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...):
![system_info](./docs/assets/system_info.png)
[armips](https://github.com/Kingcom/armips) [armips](https://github.com/Kingcom/armips)
* Supports doing bare-metal assembly for the ps1 * Supports doing bare-metal assembly for the ps1
* `scoop install armips` or just clone and build.. * `scoop install armips` or just clone and build..
* Was used early in the course. Now I just use an macro asm dsl in C11.
[luajit-2.1](https://github.com/LuaJIT/LuaJIT.git) [luajit-2.1](https://github.com/LuaJIT/LuaJIT.git)
@@ -73,3 +73,9 @@ scoop install luajit
![hello_psyq!](./docs/assets/pcsx-redux_2025-08-05_23-01-19.png) ![hello_psyq!](./docs/assets/pcsx-redux_2025-08-05_23-01-19.png)
![cube!](./docs/assets/pcsx-redux_2025-10-11_03-04-01.png) ![cube!](./docs/assets/pcsx-redux_2025-10-11_03-04-01.png)
![cube and floor!](./docs/assets/pcsx-redux_2026-07-10_22-47-02.png) ![cube and floor!](./docs/assets/pcsx-redux_2026-07-10_22-47-02.png)
Win 11 machine:
![system_info](./docs/assets/system_info.png)
Still haven't gotten around to trying this on linux...
+114 -21
View File
@@ -180,29 +180,18 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map) $link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
$link_args += ($f_link_pass_through_prefix + $f_link_start_group) $link_args += ($f_link_pass_through_prefix + $f_link_start_group)
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim.
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math,
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but
# ZERO .o files pulled in — they were unused. The 5 kept libraries
# (api, c, etc, gpu, gte) are required by the C-side calls in
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
$libraries = @( $libraries = @(
"api", "api",
"c", "c",
"c2",
"card",
"cd",
"comb",
"ds",
"etc", "etc",
"gpu", "gpu",
"gs", "gte"
"gte",
"gun",
"hmd",
"math",
"mcrd",
"mcx",
"pad",
"press",
"sio",
"snd",
"spu",
"tap"
) )
foreach ($lib in $libraries) { foreach ($lib in $libraries) {
$link_args += ($f_link_lib + $lib) $link_args += ($f_link_lib + $lib)
@@ -350,10 +339,10 @@ function build-graphis_hello {
} }
# build-graphis_hello # build-graphis_hello
function build-gte_hello { function build-hello_gte {
$includes += @() $includes += @()
$path_module = join-path $path_code 'gte_hello' $path_module = join-path $path_code 'hello_gte'
$path_duffle = join-path $path_code 'duffle' $path_duffle = join-path $path_code 'duffle'
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h' $path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
$path_build_gen = join-path $path_build 'gen' $path_build_gen = join-path $path_build 'gen'
@@ -458,8 +447,112 @@ function build-gte_hello {
} }
} }
} }
build-gte_hello # build-hello_gte
function build-hello_joypad {
$includes += @()
$path_module = join-path $path_code 'hello_joypad'
$path_duffle = join-path $path_code 'duffle'
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
$path_build_gen = join-path $path_build 'gen'
$src_c = join-path $path_module 'hello_joypad.c'
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
$assemble_args = @()
$assemble_args += $f_debug
$assemble_args += $f_optimize_none
$assemble_args += ($f_include + $path_code)
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
$module_asm_crt = join-path $path_build 'crt0.o'
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
$module_c = join-path $path_build 'hello_joypad_c.o'
$compile_args = @()
$compile_args += $f_debug
$compile_args += $f_optimize_none
# $compile_args += $f_optimize_intrinsics
# $compile_args += $f_optimize_size
# $compile_args += $f_optimize_debug
$compile_args += ($f_include + $path_code)
compile-unit $src_c $module_c $includes $compile_args
$elf = join-path $path_build 'hello_joypad.elf'
$exe = join-path $path_build 'hello_joypad.ps-exe'
$link_args = @()
$link_args += $f_debug
# $link_args += $f_optimize_size
$link_modules = @(
$module_asm_crt,
$module_c
)
link-modules $link_modules $elf $link_args
make-binary $elf $exe
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
$dwarfLineBin = join-path $path_build_gen 'hello_joypad.dwarf_line.bin'
$dwarfArangesBin = join-path $path_build_gen 'hello_joypad.dwarf_aranges.bin'
$dwarfRnglistsBin = join-path $path_build_gen 'hello_joypad.dwarf_rnglists.bin'
$injectElf = join-path $path_build 'hello_joypad.dwarf-injected.elf'
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
{
Write-Host "[build] DWARF-injecting $elf -> $injectElf"
Copy-Item -LiteralPath $elf -Destination $injectElf -Force
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
$f_args = @(
"--update-section=.debug_line=$dwarfLineBin",
"--update-section=.debug_aranges=$dwarfArangesBin",
"--update-section=.debug_rnglists=$dwarfRnglistsBin"
)
& $Objcopy @f_args $injectElf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] objcopy F' splice failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
return;
}
$dwarfInfoBin = join-path $path_build_gen 'hello_joypad.dwarf_info.bin'
$dwarfAbbrevBin = join-path $path_build_gen 'hello_joypad.dwarf_abbrev.bin'
$dwarfStrBin = join-path $path_build_gen 'hello_joypad.dwarf_str.bin'
$dwarfLocBin = join-path $path_build_gen 'hello_joypad.dwarf_loc.bin'
$dwarfLoclistsBin = join-path $path_build_gen 'hello_joypad.dwarf_loclists.bin'
$g_args = @(
"--update-section=.debug_info=$dwarfInfoBin",
"--update-section=.debug_abbrev=$dwarfAbbrevBin",
"--update-section=.debug_str=$dwarfStrBin",
"--add-section=.debug_loc=$dwarfLocBin",
"--add-section=.debug_loclists=$dwarfLoclistsBin"
)
& $Objcopy @g_args $injectElf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] objcopy G' splice failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
return;
}
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
# The original ELF and PS-EXE remain byte/flag unchanged.
& $Objcopy `
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
--set-section-flags ".data=alloc,load,data,code,contents" `
$injectElf 2>&1 | Out-Null
if ($LASTEXITCODE -ne 0) {
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
}
else {
Write-Host "[build] DWARF-injected ELF: $injectElf"
}
}
}
build-hello_joypad
# NO idea if this works yet... # NO idea if this works yet...
function Send-ToEmulator { param( [string]$exePath ) function Send-ToEmulator { param( [string]$exePath )
+54 -67
View File
@@ -920,18 +920,26 @@ function M.tokenize_body(body)
while scan <= len do while scan <= len do
local c = body:byte(scan) local c = body:byte(scan)
-- Terminator bytes (delimit a token at the top level): ',' = 0x2C, '\n' = 0x0A, ';' = 0x3B. -- Terminator bytes (delimit a token at the top level): ',' = 0x2C, '\n' = 0x0A, ';' = 0x3B.
-- These also appear as separators between argument lists inside the parens/braces/brackets, -- These also appear as separators between argument lists inside the parens/braces/brackets,
-- so we stop the scan when we hit any of them. -- so we stop the scan when we hit any of them.
if c == BYTE_COMMA then break end if c == BYTE_COMMA then break end
if c == BYTE_NEWLINE then break end if c == BYTE_NEWLINE then break end
if c == BYTE_SEMI then break end if c == BYTE_SEMI then break end
-- Line-comment '// ... \n' (0x2F 0x2F): skip to (and past) the next newline, or to end-of-body.
if c == BYTE_SLASH and body:byte(scan + 1) == BYTE_SLASH then
local nl = M.find_byte(body, BYTE_NEWLINE, scan)
scan = nl and (nl + 1) or (len + 1)
-- Block-comment '/* ... */' (0x2F 0x2A): skip to (and past) the matching '*/', or to end-of-body.
elseif c == BYTE_SLASH and body:byte(scan + 1) == BYTE_STAR then
local close = body:find("*/", scan + 2, true)
scan = close and (close + 2) or (len + 1)
-- Group opener bytes (consume the balanced group via the matching reader): '(' = 0x28, '{' = 0x7B, '[' = 0x5B. -- Group opener bytes (consume the balanced group via the matching reader): '(' = 0x28, '{' = 0x7B, '[' = 0x5B.
if c == BYTE_OPEN_PAREN then local _, a = M.read_parens (body, scan); scan = a elseif c == BYTE_OPEN_PAREN then local _, a = M.read_parens (body, scan); scan = a
elseif c == BYTE_OPEN_BRACE then local _, a = M.read_braces (body, scan); scan = a elseif c == BYTE_OPEN_BRACE then local _, a = M.read_braces (body, scan); scan = a
elseif c == BYTE_OPEN_BRACK then local _, a = M.read_brackets (body, scan); scan = a elseif c == BYTE_OPEN_BRACK then local _, a = M.read_brackets (body, scan); scan = a
-- String-literal byte ('"' = 0x22 or '\'' = 0x27): skip past the quoted region in one shot. -- String-literal byte ('"' = 0x22 or '\'' = 0x27): skip past the quoted region in one shot.
elseif c == BYTE_DQUOTE or c == BYTE_SQUOTE then elseif c == BYTE_DQUOTE or c == BYTE_SQUOTE then
scan = M.skip_str_or_cmt(body, scan) + 1 scan = (M.skip_str_or_cmt(body, scan) or scan) + 1
else else
scan = scan + 1 scan = scan + 1
end end
@@ -1180,10 +1188,9 @@ M.GTE_COMMAND_INPUTS = {
-- * "mac_result" : generic MAC output (nclip, op, mvmva) -- * "mac_result" : generic MAC output (nclip, op, mvmva)
-- --
-- Consumers: -- Consumers:
-- * passes/static_analysis.lua::analyze_hardware_relations (the walker reads this after a GTE command to update -- * passes/static_analysis.lua::analyze_hardware_relations (the walker reads this after a GTE command to update `forward_state.post_command_roles` for `gte_role_mismatch`).
-- `forward_state.post_command_roles` for `gte_result_position`). -- * passes/static_analysis.lua::check_gte_role_mismatch (per-atom CHECK_RULES reader; renders role mismatches).
-- * passes/static_analysis.lua::check_gte_result_position (per-atom CHECK_RULES reader; renders role mismatches). -- This table is consumed by the hardware-relation analyzer and the gte_role_mismatch check.
-- This table is consumed by the hardware-relation analyzer and result-position check.
M.GTE_COMMAND_OUTPUTS = { M.GTE_COMMAND_OUTPUTS = {
-- RTPS: writes one screen coordinate (the perspective-divide result) into C2_SXY2. -- RTPS: writes one screen coordinate (the perspective-divide result) into C2_SXY2.
-- The FIFO side effects leave SXY0 / SXY1 untouched, so `latest_screen_xy` is C2_SXY2. -- The FIFO side effects leave SXY0 / SXY1 untouched, so `latest_screen_xy` is C2_SXY2.
@@ -1293,32 +1300,8 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
}, },
} }
-- GTE component result contracts (immutable; keyed by bare component name). -- GTE component result contracts were removed: the `_post_<cmd>` naming convention was a soft convention
-- -- (the user did not want it formalized via static-analysis enforcement). A proper `atom_info` directive for ordering semantics is a future TODO.
-- Register-role claims that the `_post_<cmd>` suffix alone cannot infer live here.
-- The bare name (the component name stripped of the `_post_<cmd>` suffix) is the key; the row carries the expected
-- command, the expected role, and the expected C2 register.
--
-- Known rows:
-- * `gte_store_g4_p3_post_rtps`: post-RTPS polygon-emit slot reads the newest projected screen coordinate from C2_SXY2.
-- C2_SXY0 is wrong (C2_SXY0 is an older FIFO entry, never the newest post-RTPS result).
--
-- Unknown `_post_<cmd>` components (a `<name>_post_<cmd>`-suffixed component whose bare `<name>` is not a row key) emit one
-- `table_gap` info finding so downstream consumers can detect when the contract table is incomplete for an authored atom body.
--
-- Consumers:
-- * passes/static_analysis.lua::check_gte_result_position (renders result-position findings).
-- * passes/static_analysis.lua::emit_table_gap_warning (called once per atom body; surfaces the missing-row diagnostic).
-- This table is consumed by the result-position check.
M.GTE_COMPONENT_RESULT_CONTRACTS = {
-- Post-RTPS g4 p3 store contract: writes the latest screen XY (C2_SXY2) into the primitive's p3 slot.
-- Reading from C2_SXY0 is a semantic mismatch — C2_SXY0 is the oldest post-RTPS SXY, not the newest one.
["gte_store_g4_p3_post_rtps"] = {
command = "gte_cmdw_rtps",
role = "latest_screen_xy",
register = "C2_SXY2",
},
}
-- Operand-class table for the COP2->GPR load-delay check. -- Operand-class table for the COP2->GPR load-delay check.
-- --
@@ -1341,13 +1324,13 @@ M.OPERAND_READ_POSITIONS = {
["sub_s"] = {1, 2, 3}, ["sub_s"] = {1, 2, 3},
["sub_u"] = {1, 2, 3}, ["sub_u"] = {1, 2, 3},
["and_i"] = {1, 2}, ["and_i"] = {1, 2},
["and_u"] = {1, 2, 3}, ["and"] = {1, 2, 3},
["or_i"] = {1, 2}, ["or_i"] = {1, 2},
["or_i_self"] = {1}, ["or_i_self"] = {1},
["or_u"] = {1, 2, 3}, ["or"] = {1, 2, 3},
["or_u_self"] = {1, 2}, ["or_self"] = {1, 2},
["xor_i"] = {1, 2}, ["xor_i"] = {1, 2},
["xor_u"] = {1, 2, 3}, ["xor"] = {1, 2, 3},
["slt_s"] = {1, 2, 3}, ["slt_s"] = {1, 2, 3},
["slt_u"] = {1, 2, 3}, ["slt_u"] = {1, 2, 3},
["slt_si"] = {1, 2}, ["slt_si"] = {1, 2},
@@ -1447,20 +1430,21 @@ M.GP0_CMD_BY_SHAPE = {
["g4"] = 0x38, ["gt4"] = 0x3C, ["g4"] = 0x38, ["gt4"] = 0x3C,
} }
-- TODO(Ed): REMOVE THIS HARDCODE, THIS SHOULD BE RESOLVED AUTOMATICALLY
-- Per-macro prim-buffer contribution: how many 32-bit words each macro writes to the primitive being built in main RAM. -- Per-macro prim-buffer contribution: how many 32-bit words each macro writes to the primitive being built in main RAM.
-- (This counts RAM-side prim-buffer words, not .text instruction words.) -- (This counts RAM-side prim-buffer words, not .text instruction words.)
-- The sum across `mac_format_X_color` + `mac_gte_store_X_post_*` + `mac_insert_ot_tag_X` calls in an atom body must equal -- The sum across `mac_format_X_color` + `mac_gte_store_X_post_*` + `mac_insert_ot_tag_X` calls in an atom body must equal
-- `GP0_CMD_SIZE[GP0_CMD_BY_SHAPE[shape]]`. -- `GP0_CMD_SIZE[GP0_CMD_BY_SHAPE[shape]]`.
M.GP0_MACRO_CONTRIB = { M.GP0_MACRO_CONTRIB = {
["mac_format_f3_color"] = 1, ["mac_format_f3_color"] = 1,
["mac_format_g3_color"] = 3, ["mac_format_g3_color"] = 3,
["mac_format_g4_color"] = 4, ["mac_format_g4_color"] = 4,
["mac_gte_store_f3_post_rtpt"] = 3, ["mac_gte_store_f3"] = 3,
["mac_gte_store_g3_post_rtpt"] = 3, ["mac_gte_store_g3"] = 3,
["mac_gte_store_g4_p012_post_rtpt_pre_rtps"] = 3, ["mac_gte_store_g4_p012"] = 3,
["mac_gte_store_g4_p3_post_rtps"] = 1, ["mac_gte_store_g4_p3"] = 1,
["mac_insert_ot_tag_f3"] = 1, ["mac_insert_ot_tag_f3"] = 1,
["mac_insert_ot_tag_g4"] = 1, ["mac_insert_ot_tag_g4"] = 1,
} }
-- Per-macro cycle cost (best-case, no stalls). Used by the static-analysis pass to emit per-atom cycle budgets. -- Per-macro cycle cost (best-case, no stalls). Used by the static-analysis pass to emit per-atom cycle budgets.
@@ -1493,14 +1477,14 @@ M.INSTRUCTION_LATENCY = {
-- CPU ALU (single-cycle R3000A ops) -- CPU ALU (single-cycle R3000A ops)
["nop"] = 1, ["nop"] = 1,
["nop2"] = 2, ["nop2"] = 2,
["add_ui"] = 1, ["add_ui_self"] = 1, ["add_ui"] = 1, ["add_ui_self"] = 1,
["add_s"] = 1, ["add_si"] = 1, ["add_s"] = 1, ["add_si"] = 1,
["add_u"] = 1, ["add_u_self"] = 1, ["add_u"] = 1, ["add_u_self"] = 1,
["sub_u"] = 1, ["sub_s"] = 1, ["sub_u"] = 1, ["sub_s"] = 1,
["and_i"] = 1, ["and_u"] = 1, ["and_i"] = 1, ["and"] = 1,
["or_i"] = 1, ["or_i_self"] = 1, ["or_i"] = 1, ["or_i_self"] = 1,
["or_u"] = 1, ["or_u_self"] = 1, ["or_u"] = 1, ["or_u_self"] = 1,
["xor_i"] = 1, ["xor_u"] = 1, ["xor_i"] = 1, ["xor_u"] = 1,
["nor_u"] = 1, ["nor_u"] = 1,
["shift_lleft"] = 1, ["shift_lleft_self"] = 1, ["shift_lleft"] = 1, ["shift_lleft_self"] = 1,
["shift_lright"] = 1, ["shift_lright"] = 1,
@@ -1583,20 +1567,23 @@ M.INSTRUCTION_LATENCY = {
["gte_load_v1"] = 2, ["gte_load_v1"] = 2,
["gte_load_v2"] = 2, ["gte_load_v2"] = 2,
["gte_load_v0v1v2"] = 6, ["gte_load_v0v1v2"] = 6,
-- TODO(Ed): REMOVE THIS HARDCODE, THIS SHOULD BE RESOLVED AUTOMATICALLY
-- mac_* helpers (cycle cost = sum of the expanded instructions) -- mac_* helpers (cycle cost = sum of the expanded instructions)
-- mac_yield transfers control; cycle budget is 0 (the next atom absorbs the cost). -- mac_yield transfers control; cycle budget is 0 (the next atom absorbs the cost).
["mac_yield"] = 0, ["mac_yield"] = 0,
["mac_pack_color_word"] = 3, -- lui + ori + sw ["mac_pack_color_word"] = 3, -- lui + ori + sw
["mac_format_f3_color"] = 3, -- = mac_pack_color_word ["mac_format_f3_color"] = 3, -- = mac_pack_color_word
["mac_format_g4_color"] = 12, -- 4 x mac_pack_color_word ["mac_format_g4_color"] = 12, -- 4 x mac_pack_color_word
["mac_load_tri_indices"] = 3, -- 3 x lhu ["mac_load_tri_indices"] = 3, -- 3 x lhu
["mac_gte_load_tri_verts"] = 18, -- 3 x {sll, addu, lw, lw, mtc2, mtc2} ["mac_gte_load_tri_verts"] = 18, -- 3 x {sll, addu, lw, lw, mtc2, mtc2}
["mac_gte_store_f3_post_rtpt"] = 3, ["mac_gte_store_f3"] = 3,
["mac_gte_store_g3_post_rtpt"] = 3, ["mac_gte_store_g3"] = 3,
["mac_gte_store_g4_p012_post_rtpt_pre_rtps"] = 3, ["mac_gte_store_g4_p012"] = 3,
["mac_gte_store_g4_p3_post_rtps"] = 1, ["mac_gte_store_g4_p3"] = 1,
["mac_insert_ot_tag_f3"] = 11, -- 11 .word slots in the macro body ["mac_insert_ot_tag_f3"] = 11, -- 11 .word slots in the macro body
["mac_insert_ot_tag_g4"] = 11, ["mac_insert_ot_tag_g4"] = 11,
-- Annotation markers (emit no code; pure metaprogram hints) -- Annotation markers (emit no code; pure metaprogram hints)
["atom_label"] = 0, ["atom_label"] = 0,
["atom_offset"] = 0, ["atom_offset"] = 0,
@@ -1973,7 +1960,7 @@ M.GPR_VALUE_RULES = {
-- Present register-form self variants. They are included here so a -- Present register-form self variants. They are included here so a
-- known value is not needlessly lost when these encoders are used. -- known value is not needlessly lost when these encoders are used.
add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, }, add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, },
or_u_self = { op = "or_u", dest = 1, sources = {1, 2}, }, or_u_self = { op = "or", dest = 1, sources = {1, 2}, },
shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, }, shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, },
} }
+262 -1
View File
@@ -193,7 +193,7 @@ M.DWARF_LINE_OPS = {
DW_LNE_set_address = 2, -- spec: §6.2.5.3 DW_LNE_set_address = 2, -- spec: §6.2.5.3
-- Standard opcode header (§6.2.5.1) -- Standard opcode header (§6.2.5.1)
-- opcode_base + line_range are 1-byte header fields; hex so they map -- opcode_base + line_range are 1-byte header fields; hex so they map
-- directly to their position in the line-program header byte sequence. -- directly to the line-program header byte sequence.
-- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics. -- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics.
opcode_base = 0x0D, opcode_base = 0x0D,
line_base = -5, line_base = -5,
@@ -204,6 +204,44 @@ M.DWARF_LINE_OPS = {
set_address_payload_size = 0x05, -- size = sub_opcode(1) + addr(4) set_address_payload_size = 0x05, -- size = sub_opcode(1) + addr(4)
} }
-- ----------------------------------------------------------------------------
-- DWARF5 .debug_line (per DWARF5 spec §6.2.4 — Line Number Program Header)
-- ----------------------------------------------------------------------------
-- All offsets are zero-based wire offsets from the start of the unit body
-- (i.e. AFTER unit_length has been read and unit_length bytes skipped past unit_length's 4 bytes).
--
-- The DWARF3/4 line-program format differs:
-- - It omits `address_size` (DWARF3 §6.2.4) + `segment_selector_size` (DWARF5 §6.2.4).
-- - It uses null-terminated string lists for `include_directories` + `file_names`
-- (vs. DWARF5's format_count + fields-list shape).
-- These are documented inline at each parse site in read_line_unit_file_table below.
--- spec: DWARF5 spec §6.2.4 (Line Number Program Header — version >= 5)
M.DWARF5_DEBUG_LINE = {
-- Header fields (zero-based, AFTER unit_length has been read).
version_offset_post_il = 0x00, -- 2-byte LE; expected = 5
addr_size_offset = 0x02, -- 1 byte; expected = 4
seg_size_offset = 0x03, -- 1 byte; expected = 0
header_length_offset = 0x04, -- 4-byte LE; length of program-header content that follows
program_header_start = 0x08, -- first byte of program-header content (after the 8 fixed bytes)
-- Per-form byte widths (used when reading directory / file-name entries).
form_addr_bytes = 0x04, -- DW_FORM_addr (32-bit) | DW_FORM_data4
form_strp_bytes = 0x04, -- DW_FORM_line_strp / DW_FORM_strp / DW_FORM_strp_sup
form_data16_bytes = 0x10, -- DW_FORM_data16 (MD5)
-- DWARF5 form codes (subset used in line-program directory + file tables).
form_line_strp = 0x1A, -- DWARF5 §7.5.6 — DW_FORM_line_strp (4-byte offset into .debug_line_str)
form_string = 0x08, -- DWARF4-compatible fallback (inline null-terminated; not in .debug_line_str)
form_udata = 0x0F, -- DW_FORM_udata (ULEB)
form_data16 = 0x18, -- DW_FORM_data16 (16-byte MD5; gcc emits this for split debug info)
-- DWARF5 content-tag codes (DW_LNCT_* from §6.2.4.1 + §6.2.4.2).
lnct_path = 0x01,
lnct_directory_index = 0x02,
lnct_md5 = 0x05, -- gcc with MD5 in file name table (rare)
}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- I/O helpers: little-endian byte read/write -- I/O helpers: little-endian byte read/write
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -787,6 +825,229 @@ function M.sleb128_size(n)
return bytes return bytes
end end
-- ════════════════════════════════════════════════════════════════════════════
-- DWARF5 line-program file-table reader
-- ════════════════════════════════════════════
--- Read every line-program unit in `.debug_line` and produce one entry per file across all units.
--- Returns three parallel maps keyed by 1-based file index.
---
--- Wire format notes:
--- * The `.debug_line` section may contain MULTIPLE line-program units
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index`
--- emits 2 forms: path + dir_index). The helper supports:
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
--- - DW_FORM_udata (ULEB128)
--- - DW_FORM_data16 (16-byte MD5; ignored — skip the form's bytes)
--- * Symlink-canonicalisation: each path's `paths[i]` is stored verbatim from the wire
--- (mixed `/` and `\` accepted; the basename is taken via the last path separator). Caller normalises as needed.
---
--- Behavior on failure: writes to stderr and returns nil.
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
--- downstream `resolve_provenance_file_index(path)` consumers
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
--- consult the map directly.
---
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice;
--- both shapes work since the splice preserves `.debug_line`)
--- @return table|nil, table|nil, table|nil
--- basename_to_index: { [basename] = 1-based-per-unit-file-index, ... }
--- basenames: { [1-based-per-unit-file-index] = basename, ... }
--- paths: { [1-based-per-unit-file-index] = full path (mixed slashes), ... }
function M.read_line_unit_file_table(elf_path)
local sections = M.read_elf_sections(elf_path, { ".debug_line", ".debug_line_str" })
local line = sections[".debug_line"]
local lstr = sections[".debug_line_str"] or ""
if not line or line == "" then
io.stderr:write("[elf_dwarf.read_line_unit_file_table] no .debug_line section in: " .. tostring(elf_path) .. "\n")
return nil
end
local basenames = {}
local basename_to_index = {}
local paths = {}
--- Read one form-code's bytes from `buf` at position `p` according to `form`.
--- Returns (value, after) where `value` is:
--- * the resolved string (DW_FORM_line_strp / DW_FORM_string)
--- * the ULEB128 number (DW_FORM_udata)
--- * nil + skip-bytes (DW_FORM_data16; we don't surface the MD5)
local function read_form(buf, lstr_buf, p, form)
if form == M.DWARF5_DEBUG_LINE.form_line_strp then
local strp = M.read_u32_le(buf, p)
local end_pos = lstr_buf:find("\0", strp + 1, true) or (#lstr_buf + 1)
return lstr_buf:sub(strp + 1, end_pos - 1), p + M.DWARF5_DEBUG_LINE.form_strp_bytes
elseif form == M.DWARF5_DEBUG_LINE.form_string then
local nul = buf:find("\0", p + 1, true) or (#buf + 1)
return buf:sub(p + 1, nul - 1), nul
elseif form == M.DWARF5_DEBUG_LINE.form_udata then
local v, after = M.read_uleb128_at(buf, p)
return v, after
elseif form == M.DWARF5_DEBUG_LINE.form_data16 then
return nil, p + M.DWARF5_DEBUG_LINE.form_data16_bytes
else
-- Unsupported form in a directory/file-table entry: best-effort skip.
-- We do NOT stderr-write because the crt0.s DWARF5 line unit (gcc-as emitted) uses DW_FORM_addr (0x01) for what is effectively a path entry,
-- which is non-standard.
-- The C-unit's DWARF3 paths are read via the parallel DWARF3 path and never see this error.
-- Callers should consult `basename_to_index` for the paths they care about and ignore this unit if it produced none.
return nil, p
end
end
--- Parse one DWARF-version-3-style unit (DWARF3/4 line program; gcc default in the PS1 toolchain still emits DWARF3 for line programs in `-g` mode).
--- Layout: null-terminated directory list, then path(null) + dir_idx(ULEB) + time(ULEB) + size(ULEB) file entries terminated by an empty null.
--- `content_start` = zero-based wire offset of the first byte of program-header content (after version + header_length fields).
--- @return unit_basenames { [idx_in_unit_1_based] = basename }
--- @return unit_paths { [idx_in_unit_1_based] = full path }
local function parse_dwarf3_unit(buf, content_start, body_end)
local up = content_start
-- 5 fixed bytes: min_insn, default_is, line_base (signed), line_range, opcode_base
up = up + 5
local opcode_base = buf:byte(content_start + 5)
up = up + (opcode_base - 1) -- std_opcode_lengths
local dirs = {}
while up < body_end do
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
if nul > body_end then break end
local len = nul - up - 1
if len == 0 then up = nul break end
dirs[#dirs + 1] = buf:sub(up + 1, nul - 1)
up = nul
end
local unit_basenames = {}
local unit_paths = {}
while up < body_end do
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
if nul > body_end or nul == up + 1 then up = nul break end
local path = buf:sub(up + 1, nul - 1)
up = nul
local didx, up_next = M.read_uleb128_at(buf, up); up = up_next
local _time, up_next2 = M.read_uleb128_at(buf, up); up = up_next2
local _size, up_next3 = M.read_uleb128_at(buf, up); up = up_next3
local idx = #unit_basenames + 1
local bs = path:match("[^/\\]+$") or path
unit_paths[idx] = path
unit_basenames[idx] = bs
dirs[1] = dirs[1] or "" -- safety: gcc emits "" sentinel dir at 0
if didx > 0 and dirs[didx] then
unit_paths[idx] = dirs[didx] .. "/" .. path
end
end
return unit_basenames, unit_paths
end
--- Parse one DWARF-version-5-style unit (DWARF5 line program; used by modern gcc with `-gdwarf-5`).
--- `content_start` is the first byte of program-header content (after the 8 fixed bytes version+addr_size+seg_size+header_length).
--- @return same shape as parse_dwarf3_unit
local function parse_dwarf5_unit(buf, lstr_buf, content_start, body_end)
local up = content_start
-- 6 fixed bytes: min_insn, max_ops_per_insn, default_is, line_base, line_range, opcode_base
up = up + 6
local opcode_base = buf:byte(content_start + 6)
up = up + (opcode_base - 1) -- std_opcode_lengths
-- directories
local dir_format_count, after = M.read_uleb128_at(buf, up); up = after
local dir_formats = {}
for i = 1, dir_format_count do
local f, a2 = M.read_uleb128_at(buf, up); up = a2
dir_formats[i] = f
end
local dir_count, a3 = M.read_uleb128_at(buf, up); up = a3
local dirs = {}
for i = 1, dir_count do
local combined = ""
for j = 1, dir_format_count do
local v, a4 = read_form(buf, lstr_buf, up, dir_formats[j])
up = a4
if j == 1 and type(v) == "string" then combined = v end
end
dirs[i] = combined
end
-- file names
local file_format_count, after2 = M.read_uleb128_at(buf, up); up = after2
local file_formats = {}
for i = 1, file_format_count do
local f, a2 = M.read_uleb128_at(buf, up); up = a2
file_formats[i] = f
end
local file_count, a3 = M.read_uleb128_at(buf, up); up = a3
local unit_basenames = {}
local unit_paths = {}
for i = 1, file_count do
local combined = ""
local didx = 0
for j = 1, file_format_count do
local v, a4 = read_form(buf, lstr_buf, up, file_formats[j])
up = a4
if j == 1 and type(v) == "string" then combined = v end
if j == 2 and type(v) == "number" then didx = v end
end
local idx = #unit_basenames + 1
local bs = combined:match("[^/\\]+$") or combined
unit_paths[idx] = combined
unit_basenames[idx] = bs
if didx > 0 and dirs[didx] then
unit_paths[idx] = dirs[didx] .. "/" .. combined
end
end
return unit_basenames, unit_paths
end
--- Walk every line-program unit in the section.
local p = 0
local section_end = #line
while p + 4 <= section_end do
local unit_length = M.read_u32_le(line, p)
if unit_length == 0xFFFFFFFF then
io.stderr:write("[elf_dwarf.read_line_unit_file_table] 64-bit DWARF (initial-length 0xFFFFFFFF); not supported\n")
return nil
end
local body_start = p + 4
local body_end = p + 4 + unit_length
if body_end > section_end then break end
local version = M.read_u16_le(line, body_start)
local unit_basenames, unit_paths
if version >= 5 then
-- DWARF5 header: version(2) + addr_size(1) + seg_size(1) + header_length(4) + content
local header_length_offset = body_start + 6 -- past version(2) + addr_size(1) + seg_size(1) - wait that's wrong; past hdr len is at +6
local content_start = body_start + 8 -- past version(2) + addr_size(1) + seg_size(1) + header_length(4)
unit_basenames, unit_paths = parse_dwarf5_unit(line, lstr, content_start, body_end)
elseif version >= 2 then
-- DWARF2/3/4 header: version(2) + header_length(4) + content
local content_start = body_start + 6 -- past version(2) + header_length(4)
unit_basenames, unit_paths = parse_dwarf3_unit(line, content_start, body_end)
else
io.stderr:write(string.format("[elf_dwarf.read_line_unit_file_table] unsupported DWARF version %d (offset 0x%x)\n", version, p))
p = body_end
goto continue
end
-- Per-unit 1-based file indices are aligned with `inv.call_file` values because the metaprogram emits `DW_LNS_set_file` with the per-unit index.
-- When multiple units are present (crt0.s + C unit), the per-unit index in each unit matches the metaprogram's intent (gcc always sets file in unit-local terms).
-- We therefore store directly without global re-indexing; the caller is responsible for knowing which unit the file-index applies to.
-- For DWARF3 (C unit is the unit that matters for atom line tables), this matches.
-- For DWARF5 (crt0.s + C unit), each carries its own per-unit file-table map;
-- the atom-side DW_LNS_set_file(N) refers to the C unit's indices, NOT crt0.s's.
-- Since the C unit is the one with full include_directories + 12 entries, we can use it directly.
for idx, bs in pairs(unit_basenames) do
basenames[idx] = bs
paths[idx] = unit_paths[idx]
basename_to_index[bs] = idx
end
p = body_end
::continue::
end
return basename_to_index, basenames, paths
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- I/O helpers: atoms source-map + native directory glob -- I/O helpers: atoms source-map + native directory glob
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
+13 -13
View File
@@ -3,16 +3,16 @@
# Wrapper for the tape-atom step-debug helpers. # Wrapper for the tape-atom step-debug helpers.
# The 9 user commands are defined here as STUBS (degraded-state messages). # The 9 user commands are defined here as STUBS (degraded-state messages).
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua` # The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gen/gdb_tape_atoms_runtime.gdb`. # (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gdb_tape_atoms_runtime.gdb`.
# Sourcing that file RE-DEFINES the commands with real implementations. # Sourcing that file RE-DEFINES the commands with real implementations.
# #
# If `build/gen/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map). # If `build/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
# The user just needs to re-run `build_psyq.ps1` to regenerate. # The user just needs to re-run `build_psyq.ps1` to regenerate.
# ── Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ── # ?? Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ??
define tape_atoms define tape_atoms
echo "[gdb_tape_atoms] STUB: runtime file build/gen/gdb_tape_atoms_runtime.gdb not found." echo "[gdb_tape_atoms] STUB: runtime file build/gdb_tape_atoms_runtime.gdb not found."
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file." echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
end end
document tape_atoms document tape_atoms
@@ -21,35 +21,35 @@ document tape_atoms
end end
define break_atom define break_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document break_atom document break_atom
Set a breakpoint at the start of tape atom <name>. STUB state. Set a breakpoint at the start of tape atom <name>. STUB state.
end end
define step_atom define step_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document step_atom document step_atom
Resume execution until the next atom boundary. STUB state. Resume execution until the next atom boundary. STUB state.
end end
define next_atom define next_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document next_atom document next_atom
Alias for step_atom. STUB state. Alias for step_atom. STUB state.
end end
define where_in_atom define where_in_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document where_in_atom document where_in_atom
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state. Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
end end
define stepi_inside_atom define stepi_inside_atom
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1." echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
end end
document stepi_inside_atom document stepi_inside_atom
One MIPS-instruction step, then where_in_atom. STUB state. One MIPS-instruction step, then where_in_atom. STUB state.
@@ -89,17 +89,17 @@ document wave_ctx
end end
# ── Source the runtime file (re-defines commands with real impls + data). ── # ?? Source the runtime file (re-defines commands with real impls + data). ??
# Try to source from project-root-relative path first (the typical case). # Try to source from project-root-relative path first (the typical case).
# If the user is in a different CWD, the source will fail and stubs remain. # If the user is in a different CWD, the source will fail and stubs remain.
# The runtime file path is computed relative to the ELF's source map convention (build/gen/gdb_tape_atoms_runtime.gdb). # The runtime file path is computed relative to the ELF's source map convention (build/gdb_tape_atoms_runtime.gdb).
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file... echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above. # Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
# The runtime's `define` blocks are intended to overwrite there's no ambiguity to confirm. # The runtime's `define` blocks are intended to overwrite ? there's no ambiguity to confirm.
set confirm off set confirm off
# Source the runtime file (re-defines commands with real impls + data). # Source the runtime file (re-defines commands with real impls + data).
source build/gen/gdb_tape_atoms_runtime.gdb source build/gdb_tape_atoms_runtime.gdb
set confirm on set confirm on
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations). echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
-46
View File
@@ -7,18 +7,11 @@
--- ---
--- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies. --- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
--- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message. --- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
---
--- Writes `<ctx.out_root>/<dir_basename>.errors.h` once per module, with `#error` directives for findings that the C compile surfaces.
--- `passes/report.lua` renders annotations.txt from `corpus.sources_by_dir`, re-validating each source through `M.validate()`.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
-- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale. -- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
-- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module. -- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
local write_file = duffle.write_file
local ensure_dir = duffle.ensure_dir
-- The annotation pass reads the source-derived registries from scan_source: -- The annotation pass reads the source-derived registries from scan_source:
-- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks -- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks
@@ -605,40 +598,6 @@ local function validate(ctx, src, corpus_pipe_ctx)
} }
end end
-- ════════════════════════════════════════════════════════════════════════════
-- Per-DIRECTORY (per-module) output: errors.h + annotations.txt
-- ════════════════════════════════════════════════════════════════════════════
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
--- Empty directories (no errors, no atoms) produce no file.
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
if atoms_count == 0 and #errors == 0 then
return nil
end
local out_path = ctx.out_root .. "/" .. dir_basename .. ".errors.h"
local lines = {
"// Auto-generated by ps1_meta.lua (passes/annotation.lua) — DO NOT EDIT",
string.format("// Module: %s Sources: %d", dir_basename, #sources),
"#pragma once",
"",
}
if #errors == 0 then
lines[#lines + 1] = "// annotation pass OK"
else
for _, e in ipairs(errors) do
local src_tag = ""
if e.source then
local src_name = e.source:match("([^/\\]+)$") or e.source
src_tag = src_name .. ": "
end
lines[#lines + 1] = string.format('#error "%s%s (line %d)"', src_tag, e.msg, e.line)
end
end
ensure_dir(ctx.out_root)
write_file(out_path, table.concat(lines, "\n") .. "\n")
return out_path
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M.run — orchestrator entry -- M.run — orchestrator entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -683,11 +642,6 @@ function M.run(ctx)
warnings [#warnings + 1] = { line = w.line, msg = w.msg } warnings [#warnings + 1] = { line = w.line, msg = w.msg }
end end
end end
local err_path = emit_module_errors_h(ctx, dir_basename, dir_atoms, dir_errors, dir_sources)
if err_path then
table.insert(outputs, { errors_h = err_path })
end
end end
return { outputs = outputs, errors = errors, warnings = warnings } return { outputs = outputs, errors = errors, warnings = warnings }
+83 -46
View File
@@ -4,20 +4,18 @@
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime --- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime
--- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`). --- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
--- ---
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs: one --- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
--- `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes. --- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
--- ---
--- **Two output forms** (per the workspace's per-emission-form pattern from `guide_metaprogram_ssdl.md`): --- Two output forms:
--- 1. **Sourcemap.txt form** — `<out_root>/<basename>.atoms.sourcemap.txt`. Format-version-tagged for forward-compat. --- 1. Markdown form: Handled by `passes/report.lua` (writes `<module>.atoms.md`).
--- Lives in `<out_root>/` (build/gen). Mirrors the convention used by `annotation.lua` --- The render functions `render_source_map` + `render_provenance` are exported for `report.lua` to call directly.
--- (`<out_root>/<basename>.errors.h`) and `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`).
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`. --- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
--- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`. A pure gdb command script — addresses come --- 2. `gdb_tape_atoms_runtime.gdb`: Post-link opt-in (`ctx.flags.gdb_runtime`),
--- from `nm`, the 9 user commands are static `define ... end` blocks. Emitted when `ctx.flags.gdb_runtime` is true --- so the gdb wrapper script + the generated runtime script share the same canonical location.
--- AND `ctx.flags.elf_path` points to an existing ELF. Useful for `gdb-multiarch --without-python` users --- Triggered by `--post-link` or `--gdb-runtime`.
--- (the common case on Windows MinGW builds) — `source <path>` loads it with no Python / Tcl / Guile required.
--- ---
--- **Output format** (sourcemap.txt form): --- Output forma (sourcemap.txt form):
--- ``` --- ```
--- # FORMAT_VERSION 1 --- # FORMAT_VERSION 1
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT --- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
@@ -32,8 +30,6 @@
--- ``` --- ```
--- ---
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view. --- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
---
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -458,7 +454,22 @@ local function emit_gdb_runtime(ctx)
-- Confirmation line for the source operator. -- Confirmation line for the source operator.
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path' lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb" local out_path
-- Move out of `<out_root>/gdb_tape_atoms_runtime.gdb` to `<out_root>/../gdb_tape_atoms_runtime.gdb` when the conventional `<out_root>` is `<build>/gen`
-- (any equivalent spelling — relative, absolute backslash, absolute forward-slash, trailing-separator variants).
-- This puts the gdb runtime alongside the ELF at `build/` rather than under the report subdir.
local function ends_with_gen_dir(p)
if type(p) ~= "string" then return false end
return p:match("[/\\]gen[/\\]?$") ~= nil or p == "build/gen" or p == "build\\gen"
end
if ends_with_gen_dir(ctx.out_root) then
-- Strip the trailing `/gen` segment, then write the runtime script under `build/`.
-- e.g. "C:/projects/Pikuma/ps1/build/gen" -> "C:/projects/Pikuma/ps1/build".
local parent = ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
out_path = parent .. "/gdb_tape_atoms_runtime.gdb"
else
out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
end
duffle.ensure_dir(duffle.dirname(out_path)) duffle.ensure_dir(duffle.dirname(out_path))
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n") duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched)) -- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
@@ -470,10 +481,62 @@ end
local M = {} local M = {}
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`, emit two files -- Expose the pure render functions so `report.lua` and the focused tests can call them directly without triggering the file-emit path.
--- in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt` M.render_source_map = render_source_map
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation). When `ctx.flags.gdb_runtime` M.render_provenance = render_provenance
--- is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
--- Render ONE atom's sourcemap stanza.
--- @param atom table -- atom record (must have `atom.paths` populated)
--- @return string
function M.render_atom_source_map(atom)
assert(type(atom) == "table", "render_atom_source_map: atom must be a table")
assert(type(atom.paths) == "table", "render_atom_source_map: atom.paths must be a table")
local entries, total = canonical_word_entries(atom)
local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
entry.pos, entry.line, entry.text)
end
lines[#lines + 1] = "ENDATOM"
return table.concat(lines, "\n") .. "\n"
end
--- Render ONE atom's provenance stanza — no per-file format header, no enumeration of other atoms.
---
--- `rel_path` is the source path (forward-slashes) embedded in every `CALL` line.
--- The .md caller (report.lua) is expected to derive this once per `## <source>` heading and pass it down for each atom in that source.
--- @param atom table -- atom record (must have `atom.paths` populated)
--- @param wc table -- identity alias of `corpus.word_counts`
--- @param rel_path string -- source path (forward-slashes) for `CALL` fields
--- @return string
function M.render_atom_provenance(atom, wc, rel_path)
assert(type(atom) == "table", "render_atom_provenance: atom must be a table")
assert(type(atom.paths) == "table", "render_atom_provenance: atom.paths must be a table")
assert(type(rel_path) == "string", "render_atom_provenance: rel_path must be a string")
local entries, total = canonical_word_entries(atom)
local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do
local inv = entry.invocation
local macro_count = inv and wc and wc["mac_" .. inv.component_name]
if inv and macro_count ~= nil then
lines[#lines + 1] = string.format(
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
entry.pos, rel_path, entry.line, inv.component_name,
inv.def_path or "", inv.def_line or 0, entry.body_line)
else
lines[#lines + 1] = string.format(
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
end
end
return table.concat(lines, "\n") .. "\n"
end
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
@@ -495,34 +558,8 @@ function M.run(ctx)
} }
end end
-- Always emit the text form (per-source). -- atoms.sourcemap.txt + atoms.provenance.txt content moved to report.lua via `<module>.atoms.md` markdown file.
for _, src in ipairs(corpus.source_order) do -- This pass emits only the post-link gdb_runtime artifact (see emit_gdb_runtime below).
local has_projection = false
for _, atom in ipairs((src.scan or {}).atoms or {}) do
if (atom.kind == "atom" or atom.kind == "raw_atom") and atom.paths then
has_projection = true; break
end
end
if not has_projection then
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
if atom.paths then has_projection = true; break end
end
end
if has_projection then
local basename = duffle.basename_no_ext(src.path)
-- (1) atoms.sourcemap.txt — format-1 per-word call-site map.
local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt"
local sourcemap_body = render_source_map(src)
-- (2) atoms.provenance.txt — format-1 per-word definition/body map.
local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt"
local prov_body = render_provenance(src, wc)
duffle.ensure_dir(duffle.dirname(sourcemap_path))
duffle.write_file_lf(sourcemap_path, sourcemap_body)
duffle.write_file_lf(prov_path, prov_body)
outputs[#outputs + 1] = { kind = "report", path = sourcemap_path }
outputs[#outputs + 1] = { kind = "report", path = prov_path }
end
end
-- Optionally emit the gdb-runtime form (post-link, one file per build). -- Optionally emit the gdb-runtime form (post-link, one file per build).
if ctx.flags and ctx.flags.gdb_runtime then if ctx.flags and ctx.flags.gdb_runtime then
+4 -4
View File
@@ -4,7 +4,7 @@
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward. --- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
--- ---
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations, --- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
--- then resolves the function-args string from the preceding `FI_ MipsAtom ac_X(...)` declaration via a backward walk. --- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
--- ---
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation. --- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- ---
@@ -29,7 +29,7 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Atom component declaration identifiers. -- Atom component declaration identifiers.
local ATOM_COMP_PROC = "MipsAtomComp_Proc_" local ATOM_COMP_PROC = "MipsAtomComp_Proc_"
local MIPS_ATOM = "MipsAtom" -- prefix on the function declaration that wraps an AtomComp_Proc_ local MIPS_ATOM = "Slice_MipsCode" -- prefix on the function declaration that wraps an AtomComp_Proc_
-- Component-name prefixes. -- Component-name prefixes.
local AC_PREFIX = "ac_" -- arg to MipsAtomComp_(ac_X); the X is the atom name local AC_PREFIX = "ac_" -- arg to MipsAtomComp_(ac_X); the X is the atom name
@@ -97,9 +97,9 @@ local M = {}
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found. --- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
--- ---
--- Convention: function form is --- Convention: function form is
--- `FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })` --- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens. --- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
--- We then verify the preceding context ends with `MipsAtom` --- We then verify the preceding context ends with `Slice_MipsCode`
--- (the function-decl keyword with possible qualifiers between). --- (the function-decl keyword with possible qualifiers between).
--- ---
--- @param source string --- @param source string
+111 -67
View File
@@ -71,9 +71,15 @@ local DW_LNE_set_address = DWARF_LINE_OPS.DW_LNE_set_address
local DW_RLE_end_of_list = DWARF5_RNGLISTS.end_of_list local DW_RLE_end_of_list = DWARF5_RNGLISTS.end_of_list
local DW_RLE_start_length = DWARF5_RNGLISTS.start_length local DW_RLE_start_length = DWARF5_RNGLISTS.start_length
-- File index 11 in the existing main line unit is hello_gte_tape.c. -- File-index lookup for the existing main line unit (Unit 2).
-- The injector extends that unit rather than appending an unreferenced unit. -- Populated at pass start by `init_file_index_lookup(elf_path)` from the runtime ELF (see `elf_dwarf.read_line_unit_file_table`).
local ATOM_SOURCE_FILE_INDEX = 11 -- The hardcoded indices and the `PROVENANCE_BASENAME_TO_FILE_INDEX` table that previously lived here were retired in `conductor/tracks/dwarf_file_index_lookup_20260731/`
-- (red of the
-- `TODO(Ed): Remove this HARDCODE` from line 156); the runtime lookup reads the
-- actual gcc-emitted `.debug_line` file table instead.
local _file_index_by_basename = nil -- [basename] = 1-based line-table file index
local _file_path_by_index = nil -- [1-based index] = full source path (diagnostics / future consumers)
local _default_atom_source_index = nil -- any valid index used in opaque-row fallbacks
-- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31 -- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31
-- (see collect_per_source_registries + by_alias in build_inserted_children). -- (see collect_per_source_registries + by_alias in build_inserted_children).
@@ -98,9 +104,8 @@ local ABBREV_INLINED_SUBROUTINE = 0x6C -- 108: DW_TAG_inlined_subroutine with
-- (each field transitions from tape memory to GPR at load_pc + 8 = MIPS I load-delay slot boundary). -- (each field transitions from tape memory to GPR at load_pc + 8 = MIPS I load-delay slot boundary).
local ABBREV_BIND_VAR_LOCLIST = 0x6D -- 109: DW_TAG_variable no children + DW_AT_type = ref4 + DW_AT_location = sec_offset local ABBREV_BIND_VAR_LOCLIST = 0x6D -- 109: DW_TAG_variable no children + DW_AT_type = ref4 + DW_AT_location = sec_offset
-- Typed-view pointer_type (for the synthetic V4_S2* / V3_S2* / U4* / void* chains). -- Typed-view pointer_type (for the synthetic V4_S2* / V3_S2* / U4* / void* chains).
-- MUST be a fresh abbrev code in the appended table — emitting uleb128(9) collides with GCC's -- MUST be a fresh abbrev code in the appended table — emitting uleb128(9) collides with GCC's existing abbrev 9
-- existing abbrev 9 (a pointer_type that carries DW_AT_byte_size + DW_AT_type), so gdb misparses -- (a pointer_type that carries DW_AT_byte_size + DW_AT_type), so gdb misparses our 4-byte ref4 as (byte_size, type[0..2]) and lands the cursor mid-attribute.
-- our 4-byte ref4 as (byte_size, type[0..2]) and lands the cursor mid-attribute.
local ABBREV_TYPED_VIEW_POINTER = 0x6E -- 110: DW_TAG_pointer_type no children + DW_AT_type = ref4 (typed-view / U4 / void chain) local ABBREV_TYPED_VIEW_POINTER = 0x6E -- 110: DW_TAG_pointer_type no children + DW_AT_type = ref4 (typed-view / U4 / void chain)
-- DWARF5 §7.7.3 loclist opcodes. -- DWARF5 §7.7.3 loclist opcodes.
@@ -153,44 +158,66 @@ local DW_AT_inline = 0x20 -- DWARF5 §7.7.1: DW_AT_inline (used by a
local DW_AT_decl_file = 0x3A -- DWARF5 §7.7.1: DW_AT_decl_file (1-based file index into the CU's file table) local DW_AT_decl_file = 0x3A -- DWARF5 §7.7.1: DW_AT_decl_file (1-based file index into the CU's file table)
local DW_AT_decl_line = 0x3B -- DWARF5 §7.7.1: DW_AT_decl_line local DW_AT_decl_line = 0x3B -- DWARF5 §7.7.1: DW_AT_decl_line
-- File index lookup table for the existing main line unit (Unit 2). -- Replaced the hardcoded `ATOM_SOURCE_FILE_INDEX = 11` and the `PROVENANCE_BASENAME_TO_FILE_INDEX` table below with a runtime lookup
-- Provenance paths come back with mixed slashes; we normalize to basename and look up against the line unit's actual file table. -- (`init_file_index_lookup` + `resolve_provenance_file_index`) that reads the actual `.debug_line` file table from the post-link ELF.
-- Current scope has two provenance basenames: hello_gte_tape.c (the atom's call site) and lottes_tape.h (the component definition).
-- Both live in the existing gcc-generated line unit; their 1-based indices are stable across rebuilds because the include order --- Populate the module-level file-index lookup table from the `.debug_line` section of the post-link ELF pointed at by `elf_path`.
-- in code/gte_hello/hello_gte.c determines the unit's file table. --- This MUST be called exactly once at pass start (from `M.run`) before any `resolve_provenance_file_index` invocation;
-- gcc only adds a file to the line table when it has actual line-number entries; --- downstream callers handle a nil table as "no file info available; fall back to errors".
-- headers that are pure macros/typedefs (dsl.h, memory.h, math.h, mips.h, gp.h, gte.h, etc.) never appear. ---
-- lottes_tape.h is the FIRST include that emits line entries (MipsAtomComp_ declarations), so it is the FIRST entry after the primary file. --- The lookup uses `elf_dwarf.read_line_unit_file_table` (which parses both DWARF3 and DWARF5 line-program units —
local PROVENANCE_BASENAME_TO_FILE_INDEX = { --- the crt0.s assembler-side DWARF5 unit may emit non-standard form codes for paths and is intentionally skipped).
["hello_gte_tape.c"] = ATOM_SOURCE_FILE_INDEX, -- = 11 --- @param elf_path string|nil
["lottes_tape.h"] = 2, local function init_file_index_lookup(elf_path)
} if not elf_path or elf_path == "" then return end
local b2i, _basenames, paths = elf_dwarf.read_line_unit_file_table(elf_path)
if type(b2i) ~= "table" or type(paths) ~= "table" then
io.stderr:write("[dwarf_injection] read_line_unit_file_table returned no file table for: " .. tostring(elf_path) .. "\n")
return
end
_file_index_by_basename = b2i
_file_path_by_index = paths
-- Pick any valid index for the opaque-row fallbacks at lines 466 + 570
-- (both sites legitimately want "any file index"; gdb resolves whatever index we emit to whatever that file's line happens to be).
for idx in pairs(paths) do
_default_atom_source_index = idx
break
end
end
--- Resolve an absolute provenance path to the line-unit file index used by the emitting line program. --- Resolve an absolute provenance path to the line-unit file index used by the emitting line program.
--- Normalizes mixed `/` and `\` separators to a basename and looks it up against the known file table. --- Normalizes mixed `/` and `\` separators to a basename and looks it up against the runtime-computed file table populated by `init_file_index_lookup`.
--- ---
--- Fails loudly on an unknown provenance basename: adding a new component source file requires extending --- Fails loudly on an unknown provenance basename: adding a new component source file will produce a clear error message naming the missing basename and listing the .debug_line file table contents,
--- `PROVENANCE_BASENAME_TO_FILE_INDEX` so the line-program emission contract stays explicit. --- so the user can either confirm the gcc include order, the unity-root, or the `.debug_line` file table contents.
--- Silent fallback to ATOM_SOURCE_FILE_INDEX would mask the new-file case by misattributing component rows to the atom's source file. --- Silent fallback would mask the new-file case by misattributing component rows to an arbitrary source file.
--- @param path string -- absolute provenance path (e.g. "C:/.../lottes_thttps://www.youtube.com/watch?v=ORM4yLkdKx8ape.h" or "C:\\...\\lottes_tape.h") --- @param path string -- absolute provenance path (mixed slashes accepted)
--- @return integer -- 1-based line-unit file index --- @return integer -- 1-based line-unit file index
local function resolve_provenance_file_index(path) local function resolve_provenance_file_index(path)
if _file_index_by_basename == nil then
error("[dwarf_injection] resolve_provenance_file_index called before init_file_index_lookup. "
.. "Is M.run being entered correctly (with --elf)?")
end
if path == nil or path == "" then if path == nil or path == "" then
error("[dwarf_injection] resolve_provenance_file_index: empty path") error("[dwarf_injection] resolve_provenance_file_index: empty path")
end end
-- Normalize backslashes → forward slashes (paths arrive with mixed separators from the provenance file: forward slashes from Lua's io.lines; -- Normalize backslashes → forward slashes (paths arrive with mixed separators from the provenance file).
-- backslashes if the input ever round-trips through Windows shell expansion).
local normalized = path:gsub("\\", "/") local normalized = path:gsub("\\", "/")
-- Take the last path component (the basename). -- Take the last path component (the basename).
local basename = normalized:match("([^/]+)$") or normalized local basename = normalized:match("([^/]+)$") or normalized
local idx = PROVENANCE_BASENAME_TO_FILE_INDEX[basename] local idx = _file_index_by_basename[basename]
if idx == nil then if idx ~= nil then return idx end
error(string.format( -- Last-resort exact-path match (handles paths that don't reduce to a known basename).
"[dwarf_injection] resolve_provenance_file_index: unknown provenance basename '%s' (from '%s'). " for i, p in pairs(_file_path_by_index) do
.. "Extend PROVENANCE_BASENAME_TO_FILE_INDEX in passes/dwarf_injection.lua.", if p and p:gsub("\\", "/") == normalized then return i end
basename, path))
end end
return idx -- Build an error message listing the known basenames for fast diagnostics.
local known = {}
for k in pairs(_file_index_by_basename) do known[#known + 1] = k end
table.sort(known)
error(string.format("[dwarf_injection] resolve_provenance_file_index: unknown provenance basename '%s' (from '%s'). "
.. "Known basenames in the .debug_line file table (%d): %s"
, basename, path, #known, table.concat(known, ", ")))
end end
local DW_FORM_addr = 0x01 local DW_FORM_addr = 0x01
@@ -461,7 +488,7 @@ local function build_atom_sequence(atom)
if atom.debug_skip then if atom.debug_skip then
return table.concat({ return table.concat({
set_address(atom.addr), set_address(atom.addr),
set_file(ATOM_SOURCE_FILE_INDEX), set_file(resolve_provenance_file_index(atom.src_path)),
advance_line(atom.entries[1].line - 1), advance_line(atom.entries[1].line - 1),
negate_stmt(), negate_stmt(),
copy_op(), copy_op(),
@@ -508,10 +535,9 @@ local function build_atom_sequence(atom)
-- * If the invocation's body has any NESTED invocations (parent_id == top_inv.id), the body's -- * If the invocation's body has any NESTED invocations (parent_id == top_inv.id), the body's
-- first content is the call_line of the earliest nested invocation (by start_pos). -- first content is the call_line of the earliest nested invocation (by start_pos).
-- * Otherwise (only RAW words in the body), it's the line of the first raw word = body_lines[1]. -- * Otherwise (only RAW words in the body), it's the line of the first raw word = body_lines[1].
-- This is the value the multi-row PC's body_lines[1] row must reference for source-order display: -- This is the value the multi-row PC's body_lines[1] row must reference for source-order display: `anc.body_lines[1]` is the line of the FIRST WORD
-- `anc.body_lines[1]` is the line of the FIRST WORD (which for an outer whose body starts with a -- (which for an outer whose body starts with a nested expansion is inside the inner's expansion = wrong for display purposes);
-- nested expansion is inside the inner's expansion = wrong for display purposes); `anc.body_first_line` -- `anc.body_first_line` is the body's first content line in the parent's source (= correct for display).
-- is the body's first content line in the parent's source (= correct for display).
local body_first_line_of = {} local body_first_line_of = {}
for _, top_inv in ipairs(invs) do for _, top_inv in ipairs(invs) do
local earliest_nested_call_line = nil local earliest_nested_call_line = nil
@@ -565,7 +591,7 @@ local function build_atom_sequence(atom)
parts[#parts + 1] = copy_op() parts[#parts + 1] = copy_op()
end end
local call_file_idx = ATOM_SOURCE_FILE_INDEX local call_file_idx = resolve_provenance_file_index(atom.src_path)
-- --- Atom entry (idx 1) ------------------------------------------------- -- --- Atom entry (idx 1) -------------------------------------------------
local entry_1 = atom.entries[1] local entry_1 = atom.entries[1]
@@ -574,13 +600,11 @@ local function build_atom_sequence(atom)
-- If atom entry 1 starts inside an invocation, walk the ancestry and emit a call-site row + (when applicable) -- If atom entry 1 starts inside an invocation, walk the ancestry and emit a call-site row + (when applicable)
-- a body_lines[1] row for every active ancestor. For a non-nested invocation this is just the one pair; -- a body_lines[1] row for every active ancestor. For a non-nested invocation this is just the one pair;
-- for nested invocations this emits the outer call-site + body_lines[1] rows BEFORE the inner pair so the debugger displays -- for nested invocations this emits the outer call-site + body_lines[1] rows BEFORE the inner pair so the debugger displays
-- the outer body line at the inner's first word -- the outer body line at the inner's first word (PROBLEM B fix).
-- (PROBLEM B fix).
-- --
-- A marked OUTERMOST ancestor's body_lines[1] row is suppressed at this PC (the existing full-skip -- A marked OUTERMOST ancestor's body_lines[1] row is suppressed at this PC (the existing full-skip contract is preserved for the marked outer range);
-- contract is preserved for the marked outer range); its call-site row IS still emitted as a -- Its call-site row IS still emitted as a statement.
-- statement. Marked INNER ancestors always emit their body_lines[1] row with is_stmt=false -- Marked INNER ancestors always emit their body_lines[1] row with is_stmt=false (the per-invocation `want_body = not inv.debug_skip` predicate).
-- (the per-invocation `want_body = not inv.debug_skip` predicate).
if #entry_1_ancestry == 0 then if #entry_1_ancestry == 0 then
-- RAW word at atom entry: single call-site row, always a statement target. -- RAW word at atom entry: single call-site row, always a statement target.
emit_row(call_file_idx, entry_1.line, true) emit_row(call_file_idx, entry_1.line, true)
@@ -588,9 +612,8 @@ local function build_atom_sequence(atom)
-- Atom starts in an invocation. Walk the ancestry outermost-first. -- Atom starts in an invocation. Walk the ancestry outermost-first.
-- Each ancestor emits one call-site row (statement) and one body_lines[1] row -- Each ancestor emits one call-site row (statement) and one body_lines[1] row
-- (statement iff unmarked; suppressed for marked outermost). -- (statement iff unmarked; suppressed for marked outermost).
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content -- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
-- line in the parent's source), NOT anc.body_lines[1] (= the line of the first WORD, -- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call).
-- which is wrong when the outer's body starts with a nested call).
for ai, anc in ipairs(entry_1_ancestry) do for ai, anc in ipairs(entry_1_ancestry) do
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model") assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
assert(anc.body_lines[1] ~= nil assert(anc.body_lines[1] ~= nil
@@ -615,20 +638,18 @@ local function build_atom_sequence(atom)
if inv and idx == inv.start_pos + 1 then if inv and idx == inv.start_pos + 1 then
-- First word of the innermost active invocation (PROBLEM B fix — nested-display rule). -- First word of the innermost active invocation (PROBLEM B fix — nested-display rule).
-- Walk the active ancestry outermost-first; for each ancestor emit a call-site row -- Walk the active ancestry outermost-first; for each ancestor emit a call-site row (statement) + a body_lines[1] row.
-- (statement) + a body_lines[1] row. The inner-most invocation's call-site + body pair -- The inner-most invocation's call-site + body pair become the LAST two rows in the sequence.
-- become the LAST two rows in the sequence. Marked outermost ancestors suppress their -- Marked outermost ancestors suppress their body_lines[1] row at this PC (the existing full-skip contract is preserved for the marked outer range);
-- body_lines[1] row at this PC (the existing full-skip contract is preserved for the -- all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
-- marked outer range); all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
-- --
-- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC -- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC
-- for debugger context: source-level stepping now shows the outer body line (not the -- for debugger context: source-level stepping now shows the outer body line
-- inner body line) when stepping into the inner. PROBLEM B fix. -- (not the inner body line) when stepping into the inner. PROBLEM B fix.
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content -- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
-- line in the parent's source), NOT anc.body_lines[1] (= the line of the first WORD, -- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call:
-- which is wrong when the outer's body starts with a nested call: gdb 12.1 picks the -- gdb 12.1 picks the displayed line as the LAST row at the same PC in byte-stream order,
-- displayed line as the LAST row at the same PC in byte-stream order, so the disc=1 row's -- so the disc=1 row's value matters for what's shown when stepping into the nested case).
-- value matters for what's shown when stepping into the nested case).
local ancestry = ancestry_idx[idx] local ancestry = ancestry_idx[idx]
for ai, anc in ipairs(ancestry) do for ai, anc in ipairs(ancestry) do
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model") assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
@@ -647,10 +668,9 @@ local function build_atom_sequence(atom)
-- Subsequent body word of the innermost active invocation: `body_lines[k]` is indexed by the 1-based offset of this word inside the invocation. -- Subsequent body word of the innermost active invocation: `body_lines[k]` is indexed by the 1-based offset of this word inside the invocation.
-- Both `idx` (1-based DWARF entry index) and `inv.start_pos` (0-based emitted-word position stamped at `emit_invoke_begin`) come from the same -- Both `idx` (1-based DWARF entry index) and `inv.start_pos` (0-based emitted-word position stamped at `emit_invoke_begin`) come from the same
-- monotonic counter, so `idx - inv.start_pos` is exactly the 1-based k (the first word of the invocation has `idx == inv.start_pos + 1`, hence `k == 1`). -- monotonic counter, so `idx - inv.start_pos` is exactly the 1-based k (the first word of the invocation has `idx == inv.start_pos + 1`, hence `k == 1`).
-- atom_dbg_step_ux_20260725: `want_body = not inv.debug_skip`. The previous `want = not marked_idx[idx]` -- atom_dbg_step_ux_20260725: `want_body = not inv.debug_skip`.
-- (which suppressed ALL body rows when any ancestor was marked) is replaced by the per-invocation -- The previous `want = not marked_idx[idx]` (which suppressed ALL body rows when any ancestor was marked) is replaced by the per-invocation predicate.
-- predicate. Marked invocations emit non-statement body rows at every body word; unmarked -- Marked invocations emit non-statement body rows at every body word; unmarked invocations emit statement body rows.
-- invocations emit statement body rows.
assert(inv.body_lines, "missing body_lines: emitter did not run emission-model") assert(inv.body_lines, "missing body_lines: emitter did not run emission-model")
local words_into = idx - inv.start_pos local words_into = idx - inv.start_pos
assert(inv.body_lines[words_into] ~= nil assert(inv.body_lines[words_into] ~= nil
@@ -704,7 +724,9 @@ local function build_atom_table(corpus, addrs)
local atoms_by_name = corpus.atoms_by_name or {} local atoms_by_name = corpus.atoms_by_name or {}
-- Per-atom ingest. Returns nil if the atom is absent from the corpus; the caller skips it via the `if atom then ...` guard. -- Per-atom ingest. Returns nil if the atom is absent from the corpus; the caller skips it via the `if atom then ...` guard.
local function ingest_atom(name, info) -- `src_path` is the absolute source path that declared this atom; the build_atom_table iteration below threads `src.path` through.
-- This is consumed by `build_atom_sequence::set_file(...)` for opaque-row fallbacks + raw-word rows (atoms where no invocation ancestry exists).
local function ingest_atom(name, info, src_path)
local atom_record = atoms_by_name[name] local atom_record = atoms_by_name[name]
if not atom_record then return nil end if not atom_record then return nil end
@@ -730,6 +752,7 @@ local function build_atom_table(corpus, addrs)
words = #word_events, words = #word_events,
entries = entries, entries = entries,
debug_skip = atom_record.debug_skip == true, debug_skip = atom_record.debug_skip == true,
src_path = src_path or "",
} }
-- Consume invocation records from `atom.paths.invocations`. It is the single producer of per-invocation body_lines, per-invocation debug_skip, -- Consume invocation records from `atom.paths.invocations`. It is the single producer of per-invocation body_lines, per-invocation debug_skip,
@@ -753,9 +776,26 @@ local function build_atom_table(corpus, addrs)
end end
local out = {} local out = {}
for name, info in pairs(addrs) do -- Walk every source's atom list (which preserves source order + per-source src_path).
local atom = ingest_atom(name, info) -- Cross-ref with the nm symbol table; atoms absent from `addrs` are skipped (an atom
if atom then out[#out + 1] = atom end -- declared in source but not emitted as a symbol is a metaprogram or atom-info bug, not
-- a source-correlation bug — emit_no_emit would catch it upstream).
for _, src in ipairs((corpus and corpus.source_order) or {}) do
local src_path = src.path or ""
for _, atom_rec in ipairs(((src.scan or {}).atoms) or {}) do
local info = addrs[atom_rec.name or atom_rec.raw_name]
if info then
local atom = ingest_atom(atom_rec.name or atom_rec.raw_name, info, src_path)
if atom then out[#out + 1] = atom end
end
end
for _, atom_rec in ipairs(((src.scan or {}).raw_atoms) or {}) do
local info = addrs[atom_rec.name or atom_rec.raw_name]
if info then
local atom = ingest_atom(atom_rec.name or atom_rec.raw_name, info, src_path)
if atom then out[#out + 1] = atom end
end
end
end end
table.sort(out, function(a, b) return a.addr < b.addr end) table.sort(out, function(a, b) return a.addr < b.addr end)
return out return out
@@ -2186,6 +2226,10 @@ function M.run(ctx)
-- reading them just returns "" which is the "missing" case the builder handles. -- reading them just returns "" which is the "missing" case the builder handles.
".debug_loc", ".debug_loclists", ".debug_loc", ".debug_loclists",
}) })
-- Resolve the per-file line-table indices from the same .debug_line bytes;
-- this MUST run before any atom sequence is emitted (build_atom_sequence below
-- calls resolve_provenance_file_index when populating call-site / body rows).
init_file_index_lookup(elf_path)
-- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation). -- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation).
-- `corpus` is the sole canonical source projection. -- `corpus` is the sole canonical source projection.
local corpus = (ctx.shared and ctx.shared.corpus) or {} local corpus = (ctx.shared and ctx.shared.corpus) or {}
+410 -287
View File
@@ -7,9 +7,6 @@
--- ---
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory. --- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results. --- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -18,17 +15,21 @@
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD. -- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
-- Bootstrap: see `ps1_meta.lua` for the rationale. -- Bootstrap: see `ps1_meta.lua` for the rationale.
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath). -- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
-- Uses `debug.getinfo` to find this file's own directory, so it works -- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
-- both standalone and when require'd from the orchestrator.
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd). -- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module. -- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./" local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua") local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection. -- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings) -- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output. -- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
local annotation = dofile(_bootstrap_dir .. "annotation.lua") local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
local atoms_source_map = dofile(_bootstrap_dir .. "atoms_source_map.lua")
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Constants -- Constants
@@ -44,8 +45,7 @@ local SECTION_HEADER_MACROS = "── Macro word-count declarations ───
local SECTION_HEADER_ERRORS = "── Errors ──────────────────────────────────────────────" local SECTION_HEADER_ERRORS = "── Errors ──────────────────────────────────────────────"
local SECTION_HEADER_WARNINGS = "── Warnings ────────────────────────────────────────────" local SECTION_HEADER_WARNINGS = "── Warnings ────────────────────────────────────────────"
-- Lua pattern that captures the basename (last path segment) of a -- Lua pattern that captures the basename (last path segment) of a forward- or back-slash separated path.
-- forward- or back-slash separated path.
local BASENAME_PATTERN = "([^/\\]+)$" local BASENAME_PATTERN = "([^/\\]+)$"
-- Debug flag name — set to truthy in `_G` to enable verbose logging. -- Debug flag name — set to truthy in `_G` to enable verbose logging.
@@ -148,328 +148,451 @@ local function source_basename(path)
return path:match(BASENAME_PATTERN) or path return path:match(BASENAME_PATTERN) or path
end end
--- (internal) Format a single annotation entry as one rendered line. -- ════════════════════════════════════════════════════════════════════════════
--- @param a AnnotEntry -- Markdown renderers (consolidated-report-files refactor, 2026-07-26)
--- @param src_name string -- ════════════════════════════════════════════════════════════════════════════
--- Render the thin project-wide summary (`build/atom_meta_report.summary.md`).
--- @param all_results { module:string, atoms:integer, annots:integer, binds:integer,
--- macros:integer, findings:integer, errors:integer,
--- warnings:integer, info:integer }[]
--- @return string --- @return string
local function format_annot_line(a, src_name) local function render_project_summary(all_results)
if a.error then local lines = {
return string.format(" ✗ line %d %s [ERROR: %s] [%s]", a.line, a.macro or "?", a.error, src_name) "# Project summary",
end "> Auto-generated by ps1_meta.lua (passes/report.lua).",
local line = string.format(" ● line %d %s [%s]", a.line, a.name, src_name) "",
if a.binds then line = line .. " binds=" .. a.binds end "| module | atoms | annots | binds | macros | findings | errors | warnings | info |",
if #a.reads > 0 then line = line .. " reads={" .. table.concat(a.reads, ",") .. "}" end "|--------|-------|--------|-------|--------|----------|--------|----------|------|",
if #a.writes > 0 then line = line .. " writes={" .. table.concat(a.writes, ",") .. "}" end
return line
end
--- (internal) Tally totals across all results in a module.
--- @param results AnnotationResult[]
--- @return integer, integer, integer, integer, integer, integer
local function tally_module_totals(results)
local total_atoms, total_annots, total_binds, total_macros = 0, 0, 0, 0
local total_errors, total_warnings = 0, 0
for _, r in ipairs(results) do
total_atoms = total_atoms + #r.atoms
total_annots = total_annots + #r.annots
total_binds = total_binds + #r.binds
total_macros = total_macros + #r.macros
total_errors = total_errors + #r.errors
total_warnings = total_warnings + #r.warnings
end
return total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings
end
-- (internal) Section renderer: per-source atom declarations.
local function render_module_atoms_section(add, results)
add(SECTION_HEADER_ATOMS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do
add(string.format(" MipsAtom_(%s) line %d [%s]", a.name, a.line, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source annotation entries.
local function render_module_annots_section(add, results)
add(SECTION_HEADER_ANNOTS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
add(format_annot_line(a, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source Binds_* struct declarations.
local function render_module_binds_section(add, results)
add(SECTION_HEADER_BINDS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format(" %s line %d %d bytes [%s]", b.name, b.line, b.bytes, src_name))
for _, f in ipairs(b.fields) do
add(string.format(" +%2d: %s", f.offset, f.name))
end
end
end
add("")
end
-- (internal) Section renderer: per-source macro word-count declarations.
local function render_module_macros_section(add, results)
add(SECTION_HEADER_MACROS)
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format(" %s line %d words=%d [%s]", m.name, m.line, m.words, src_name))
end
end
add("")
end
-- (internal) Section renderer: per-source errors (one-line + "(none)" if empty).
local function render_module_errors_section(add, results, total_errors)
add(SECTION_HEADER_ERRORS)
if total_errors == 0 then
add(" (none)")
else
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, e in ipairs(r.errors) do
add(string.format(" ✗ line %d %s [%s]", e.line, e.msg, src_name))
end
end
end
add("")
end
-- (internal) Section renderer: per-source warnings (one-line + "(none)" if empty).
local function render_module_warnings_section(add, results, total_warnings)
add(SECTION_HEADER_WARNINGS)
if total_warnings == 0 then
add(" (none)")
else
for _, r in ipairs(results) do
local src_name = source_basename(r.source)
for _, w in ipairs(r.warnings) do
add(string.format(" ⚠ line %d %s [%s]", w.line, w.msg, src_name))
end
end
end
add("")
end
-- ════════════════════════════════════════════════════════════════════════════
-- SECTION_RENDERERS — data-driven section dispatch (the plex pattern)
-- ════════════════════════════════════════════════════════════════════════════
--
-- Each entry maps a section to its (header, render_fn). The render_fn signature:
-- render_fn(add, results, totals)
-- add -- the `add(line)` closure from the surrounding report renderer
-- results -- AnnotationResult[] (per-source results)
-- totals -- {atoms, annots, binds, macros, errors, warnings} counts
--
-- Sections that need to render "(none)" vs iterate use totals.errors / totals.warnings;
-- other sections ignore the totals arg.
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
local SECTION_RENDERERS = {
{ header = SECTION_HEADER_ATOMS, render = render_module_atoms_section },
{ header = SECTION_HEADER_ANNOTS, render = render_module_annots_section },
{ header = SECTION_HEADER_BINDS, render = render_module_binds_section },
{ header = SECTION_HEADER_MACROS, render = render_module_macros_section },
{ header = SECTION_HEADER_ERRORS, render = function(add, results, totals) return render_module_errors_section(add, results, totals.errors) end },
{ header = SECTION_HEADER_WARNINGS, render = function(add, results, totals) return render_module_warnings_section(add, results, totals.warnings) end },
}
--- Render the per-MODULE annotation report (one `<dir_basename>.annotations.txt`).
--- @param dir string -- module directory path
--- @param sources SourceFile[] -- sources in this module
--- @param results AnnotationResult[] -- per-source validate() results
--- @return string -- the rendered report text
local function render_module_report(dir, sources, results)
local lines = {}
local function add(s) lines[#lines + 1] = s end
add(RULE_THICK)
add("ANNOTATION PASS — module " .. source_basename(dir))
add(RULE_THICK)
add(string.format("Sources: %d", #sources))
for _, s in ipairs(sources) do add(" " .. s.path) end
add("")
local total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings = tally_module_totals(results)
add(string.format("Atoms: %d Annotations: %d Binds structs: %d Macro decls: %d",
total_atoms, total_annots, total_binds, total_macros))
add("")
-- Bundle the totals so the section renderers don't need separate parameter lists.
-- Errors/warnings sections need their total count to decide "(none)" vs iterate.
-- Sections without totals (atoms/annots/binds/macros) ignore this arg.
local totals = {
atoms = total_atoms, annots = total_annots, binds = total_binds,
macros = total_macros, errors = total_errors, warnings = total_warnings,
} }
local totals = { atoms = 0, annots = 0, binds = 0, macros = 0,
-- THE per-section dispatch. ONE loop over SECTION_RENDERERS. findings = 0, errors = 0, warnings = 0, info = 0 }
-- Each renderer writes its header + content via the `add` closure (pre-bound above). for _, e in ipairs(all_results) do
-- Adding a new section = 1 row here + 1 render_<thing>_section function. lines[#lines + 1] = string.format(
for _, section in ipairs(SECTION_RENDERERS) do "| %s | %d | %d | %d | %d | %d | %d | %d | %d |",
section.render(add, results, totals) e.module, e.atoms, e.annots, e.binds, e.macros,
e.findings, e.errors, e.warnings, e.info)
totals.atoms = totals.atoms + e.atoms
totals.annots = totals.annots + e.annots
totals.binds = totals.binds + e.binds
totals.macros = totals.macros + e.macros
totals.findings = totals.findings + e.findings
totals.errors = totals.errors + e.errors
totals.warnings = totals.warnings + e.warnings
totals.info = totals.info + e.info
end end
lines[#lines + 1] = string.format(
"| **TOTAL** | %d | %d | %d | %d | %d | %d | %d | %d |",
totals.atoms, totals.annots, totals.binds, totals.macros,
totals.findings, totals.errors, totals.warnings, totals.info)
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
-- ════════════════════════════════════════════════════════════════════════════ --- Render the per-module verbose source-map markdown (`build/<module>.atoms.md`).
-- Per-project summary --- Per-source sub-section, per-atom stanza with sourcemap + provenance rows.
-- ════════════════════════════════════════════════════════════════════════════ --- Pulls sourcemap + provenance from `atoms_source_map` (no second source walk).
--- @param dir string
--- Render the per-project summary (`build/gen/annotation_validation.txt`). --- @param dir_sources SourceFile[]
--- Aggregates totals across all sources; lists per-source error counts if any source has errors. --- @param wc table<string, integer>
--- @param all_results AnnotationResult[]
--- @return string --- @return string
local function render_project_report(all_results) local function render_module_atoms_md(dir, dir_sources, wc)
local lines = {} local dir_basename = source_basename(dir)
local lines = {
"# " .. dir_basename .. " — atoms (verbose source map)",
"> Per-word call-site + provenance. Auto-generated.",
"",
}
for _, src in ipairs(dir_sources) do
local src_name = source_basename(src.path)
lines[#lines + 1] = "## " .. src_name
lines[#lines + 1] = ""
-- For each atom with a projection, render its sourcemap + provenance.
local atoms_list = {}
for _, atom in ipairs((src.scan or {}).atoms or {}) do
if atom.paths then atoms_list[#atoms_list + 1] = atom end
end
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
if atom.paths then atoms_list[#atoms_list + 1] = atom end
end
if #atoms_list == 0 then
lines[#lines + 1] = "_(no atom projections)_"
lines[#lines + 1] = ""
else
-- Per-source forward-slash path (same one `emit_atom_stanza` / `emit_provenance_stanza` would derive;
-- computed once per `## <source>` heading and reused by each atom's `WORD N CALL ...` field).
local rel_path = src.path:gsub("\\\\", "/")
for _, atom in ipairs(atoms_list) do
lines[#lines + 1] = string.format(
"### atom: %s (line %d, %d words)",
atom.name, atom.line or 0, #(atom.paths.items or {}))
lines[#lines + 1] = ""
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
lines[#lines + 1] = "```"
-- Per-atom invariant: call the per-atom renderers, NOT the per-source ones.
-- The per-source renderers enumerate every atom in `src`;
-- calling them in a per-atom loop would repeat the whole source under every `### atom:` heading.
lines[#lines + 1] = atoms_source_map.render_atom_source_map(atom):gsub("\n+$", "")
lines[#lines + 1] = "```"
lines[#lines + 1] = ""
lines[#lines + 1] = "**Provenance** — per-word definition + body:"
lines[#lines + 1] = "```"
lines[#lines + 1] = atoms_source_map.render_atom_provenance(atom, wc, rel_path):gsub("\n+$", "")
lines[#lines + 1] = "```"
lines[#lines + 1] = ""
end
end
end
return table.concat(lines, "\n") .. "\n"
end
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
--- Aggregates annotation + static-analysis content across all sources in `dir`.
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
--- @param dir string
--- @param dir_sources SourceFile[]
--- @param annot_results AnnotationResult[]
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
--- @return string
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
local dir_basename = source_basename(dir)
local lines = {
"# " .. dir_basename .. " — atom meta report",
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
"",
}
local function add(s) lines[#lines + 1] = s end local function add(s) lines[#lines + 1] = s end
local total_atoms, total_annots, total_macros, total_binds = 0, 0, 0, 0 -- Module summary table.
local total_errors, total_warnings = 0, 0 local n_atoms = 0
for _, r in ipairs(all_results) do local n_annot = 0
total_atoms = total_atoms + #r.atoms local n_binds = 0
total_annots = total_annots + #r.annots local n_macros = 0
total_macros = total_macros + #r.macros local n_bare, n_proc = 0, 0
total_binds = total_binds + #r.binds for _, r in ipairs(annot_results) do
total_errors = total_errors + #r.errors n_atoms = n_atoms + #r.atoms
total_warnings = total_warnings + #r.warnings n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
for _, a in ipairs(sa_results.atoms or {}) do
if a.kind == "comp_bare" then n_bare = n_bare + 1
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
end
end end
add(RULE_THICK) add("## Module summary"); add("")
add("ANNOTATION VALIDATION — project summary") add("| metric | value |"); add("|--------|-------|")
add(RULE_THICK) add(string.format("| sources | %d |", #dir_sources))
add("") add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
add(string.format("Atoms: %d", total_atoms)) #(sa_results.atoms or {}),
add(string.format("Annotations: %d", total_annots)) #(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
add(string.format("Macros: %d", total_macros)) add(string.format("| annotations | %d |", n_annot))
add(string.format("Binds: %d", total_binds)) add(string.format("| binds structs | %d |", n_binds))
add("") add(string.format("| macro decls | %d |", n_macros))
add(string.format("Errors: %d", total_errors)) add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
add(string.format("Warnings: %d", total_warnings)) #(sa_results.findings or {}),
#(sa_results.errors or {}),
#(sa_results.warnings or {}),
#(sa_results.info or {})))
add("") add("")
if total_errors > 0 then -- Sources
add("Per-source error counts:") add("## Sources"); add("")
for _, r in ipairs(all_results) do for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
if #r.errors > 0 then add("")
local src_name = source_basename(r.source)
add(string.format(" %s : %d error(s)", src_name, #r.errors)) -- Atoms (annotation)
add("## Atoms"); add("")
add("| kind | name | source | line |"); add("|------|------|--------|------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
end
end
add("")
-- Annotations
add("## Annotations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | name | binds | reads | writes |")
add("|--------|------|------|-------|-------|--------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
local binds = a.binds or ""
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or ""
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or ""
add(string.format("| %s | %d | %s | %s | %s | %s |",
src_name, a.line, a.name, binds, reads, writes))
end
end
end
add("")
-- Binds_* structs
add("## Binds_* structs"); add("")
if #annot_results == 0 then
add("_(none)_")
else
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format("### %s (%s:%d, %d bytes)",
b.name, src_name, b.line, b.bytes))
for _, f in ipairs(b.fields) do
add(string.format("- `+%d %s`", f.offset, f.name))
end
add("")
end
end
end
-- Macro decls
add("## Macro word-count declarations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | macro declaration |")
add("|--------|------|-------------------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format("| %s | %d | %s |",
src_name, m.line, m.name))
end
end
end
add("")
-- Findings by atom (static-analysis)
add("## Static analysis — findings by atom"); add("")
local by_atom = {}
for _, f in ipairs(sa_results.findings or {}) do
by_atom[f.atom] = by_atom[f.atom] or {}
by_atom[f.atom][#by_atom[f.atom] + 1] = f
end
if next(by_atom) == nil then
add("_(no findings)_")
else
for _, a in ipairs(sa_results.atoms or {}) do
local fs = by_atom[a.name]
if fs then
add(string.format("### %s", a.name))
for _, f in ipairs(fs) do
add(string.format("- `[%s] %s`", f.check, f.msg))
end
add("")
end
end
end
-- Errors / Warnings / Info
local function add_findings(label, entries)
add(string.format("## %s", label))
if #entries == 0 then
add("_(none)_")
else
for _, e in ipairs(entries) do
add(string.format("- line %d %s", e.line, e.msg))
end end
end end
add("") add("")
end end
add_findings("Errors", sa_results.errors or {})
add_findings("Warnings", sa_results.warnings or {})
add_findings("Info", sa_results.info or {})
-- Per-atom cycle counts (path-aware)
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
add("| atom | source | min | max | branches | paths | notes |")
add("|------|--------|-----|-----|----------|-------|-------|")
local sorted = {}
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
table.sort(sorted, function(x, y)
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
end)
for _, a in ipairs(sorted) do
local p = a.paths or {}
local src_name = a.source_path and source_basename(a.source_path) or ""
local notes = ""
if p.has_loops then notes = notes .. " [loop!]" end
if p.unknown_macros and #p.unknown_macros > 0 then
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
end
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
a.name, src_name,
p.cycles_min or 0, p.cycles_max or 0,
p.branches or 0, p.paths or 0, notes))
end
add("")
-- Per-source scan summary
add("## Per-source scan summary"); add("")
for _, src in ipairs(dir_sources) do
local src_atoms = {}
for _, a in ipairs(sa_results.atoms or {}) do
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
end
if #src_atoms > 0 then
local mn, mx = math.huge, -1
for _, a in ipairs(src_atoms) do
local p = a.paths or {}
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
end
local path_str
if mx > 0 then
path_str = string.format(" cycles=%d..%d", mn, mx)
else
path_str = string.format(" %d cycles", mn)
end
add(string.format("- `%s` — %d atom%s%s",
src.basename, #src_atoms,
#src_atoms == 1 and "" or "s", path_str))
end
end
add("")
return table.concat(lines, "\n") .. "\n" return table.concat(lines, "\n") .. "\n"
end end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Orchestration helpers -- REPORT_RENDERERS — data-driven report dispatch (one row per file kind)
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- `once = true` means render once at the project level (not per-module).
--- (internal) Re-validate every source in a directory against the canonical corpus projection. -- `basename(dir_basename)` yields the file's basename for that kind.
--- Calls `annotation.validate()` per source to produce the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings) -- `gather(ctx, dir, dir_sources [, all_modules])` returns the rendered string.
--- that the report renderer consumes. Eeach report pass run is reproducible from the corpus. local REPORT_RENDERERS = {
--- Returns the list of module results + the flat list of all results (for the project-wide summary). {
--- @param ctx PassCtx name = "atom_meta_report",
--- @param dir_sources SourceFile[] ext = "md",
--- @return AnnotationResult[], AnnotationResult[] basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
local function lookup_module_results(ctx, dir_sources) once = false,
local module_results = {} gather = function(ctx, dir, dir_sources)
local all_results = {} -- Annotations: re-run `annotation.validate()` per source (the existing pattern).
for _, src in ipairs(dir_sources) do local annot_results = {}
if src.scan then for _, src in ipairs(dir_sources) do
local result = annotation.validate(ctx, src, nil) if src.scan then
result.source = src.path -- tag for downstream rendering local r = annotation.validate(ctx, src, nil)
module_results[#module_results + 1] = result r.source = src.path
all_results[#all_results + 1] = result annot_results[#annot_results + 1] = r
end end
end end
return module_results, all_results -- Static-analysis: read stashed projection (no re-validate).
end local dir_basename = dir:match("([^/\\]+)$") or dir
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
--- (internal) Does this module's results contain anything worth emitting? return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
--- @param module_results AnnotationResult[] end,
--- @return boolean },
local function module_has_content(module_results) {
for _, r in ipairs(module_results) do name = "atoms",
if #r.atoms > 0 or #r.annots > 0 or #r.binds > 0 ext = "md",
or #r.macros > 0 or #r.errors > 0 or #r.warnings > 0 then basename = function(dir_basename) return dir_basename .. ".atoms" end,
return true once = false,
end gather = function(ctx, dir, dir_sources)
end return render_module_atoms_md(dir, dir_sources,
return false ctx.shared.corpus.word_counts or {})
end end,
},
--- (internal) Log a debug message if `_G[DEBUG_FLAG]` is truthy. {
--- @param fmt string name = "summary",
local function debug_log(fmt, ...) ext = "md",
if _G[DEBUG_FLAG] then basename = function(_dir_basename) return "atom_meta_report.summary" end,
io.stderr:write(string.format("[%s] " .. fmt, PASS_NAME, ...)) once = true,
end gather = function(_ctx, _dir, _dir_sources, all_modules)
end return render_project_summary(all_modules)
end,
},
}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M — module exports -- M — public pass surface
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
local M = {} local M = {}
--- Run the report pass. --- Run the report pass. Emits 1 `atom_meta_report.summary.md` per build + 2 `atom_meta_report.md` + 2 `atoms.md` files per module (duffle + gte_hello).
--- Renders one `<dir_basename>.annotations.txt` per source-directory that has content, plus the project-wide `annotation_validation.txt` summary. --- Reads `corpus.static_analysis_results` (added in Phase 1) to populate per-module findings without re-running validate().
--- @param ctx PassCtx --- @param ctx PassCtx
--- @return PassResult --- @return PassResult
function M.run(ctx) function M.run(ctx)
local outputs = {} local outputs = {}
local errors = {} local corpus = ctx.shared and ctx.shared.corpus
local warnings = {} local by_dir = (corpus and corpus.sources_by_dir) or {}
-- Module grouping comes from `corpus.sources_by_dir` (the canonical projection). -- `out_path_root`: when the conventional `out_root` is `build/gen` (any spelling — relative, absolute, separator variants).
-- Iterate it directly; no private cache, no per-pass stash. -- Write the md files to `build/` (parent of `gen/`) instead of nested under `gen/`.
local corpus = ctx.shared and ctx.shared.corpus -- Mirrors the `gdb_tape_atoms_runtime.gdb` relocation.
local by_dir = (corpus and corpus.sources_by_dir) or {} local function ends_with_gen(p)
return type(p) == "string" and (p:match("[/\\]gen[/\\]?$") ~= nil
or p == "build/gen" or p == "build\\gen")
end
local out_root_effective = ends_with_gen(ctx.out_root)
and ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
or ctx.out_root
duffle.ensure_dir(ctx.out_root) duffle.ensure_dir(out_root_effective)
-- Aggregator for the project-wide `once = true` summary renderer.
local all_modules = {}
local all_results_for_summary = {}
for dir, dir_sources in pairs(by_dir) do for dir, dir_sources in pairs(by_dir) do
local dir_basename = dir:match("([^/\\]+)$") or dir local dir_basename = dir:match("([^/\\]+)$") or dir
debug_log("dir=%s basename=%s sources=%d\n", dir, dir_basename, #dir_sources)
if #dir_sources > 0 then -- Per-renderer dispatch for the per-module renderers (once = false).
local module_results, all_results = lookup_module_results(ctx, dir_sources) for _, renderer in ipairs(REPORT_RENDERERS) do
for _, r in ipairs(all_results) do if not renderer.once then
all_results_for_summary[#all_results_for_summary + 1] = r local body = renderer.gather(ctx, dir, dir_sources)
local out_path = out_root_effective .. "/" .. renderer.basename(dir_basename) .. "." .. renderer.ext
duffle.write_file(out_path, body)
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
end end
end
if module_has_content(module_results) then -- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
local out_path = ctx.out_root .. "/" .. dir_basename .. ".annotations.txt" local annot_results = {}
duffle.write_file(out_path, render_module_report(dir, dir_sources, module_results)) for _, src in ipairs(dir_sources) do
outputs[#outputs + 1] = { annotations_txt = out_path } if src.scan then
else local r = annotation.validate(ctx, src, nil)
debug_log(" -> no content; skipping\n") r.source = src.path
annot_results[#annot_results + 1] = r
end end
end end
local n_annot, n_binds, n_macros = 0, 0, 0
for _, r in ipairs(annot_results) do
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
all_modules[#all_modules + 1] = {
module = dir_basename,
atoms = #(sa_results.atoms or {}),
annots = n_annot,
binds = n_binds,
macros = n_macros,
findings = #(sa_results.findings or {}),
errors = #(sa_results.errors or {}),
warnings = #(sa_results.warnings or {}),
info = #(sa_results.info or {}),
}
end
-- Project-wide renderer (once = true): write the summary file.
for _, renderer in ipairs(REPORT_RENDERERS) do
if renderer.once then
local body = renderer.gather(ctx, nil, nil, all_modules)
local out_path = out_root_effective .. "/" .. renderer.basename("") .. "." .. renderer.ext
duffle.write_file(out_path, body)
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
end
end end
if #all_results_for_summary > 0 then return { outputs = outputs, errors = {}, warnings = {} }
local summary_path = ctx.out_root .. "/annotation_validation.txt"
duffle.write_file(summary_path, render_project_report(all_results_for_summary))
outputs[#outputs + 1] = { summary_txt = summary_path }
end
return { outputs = outputs, errors = errors, warnings = warnings }
end end
return M return M
+118 -275
View File
@@ -39,17 +39,19 @@
--- The report header includes `Info: N` alongside Findings / Errors / Warnings, and a dedicated --- The report header includes `Info: N` alongside Findings / Errors / Warnings, and a dedicated
--- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts. --- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts.
--- ---
--- The structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`, `control_transfer_delay_slot_use`) skip atoms/components with `debug_skip == true`.
--- The `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
--- Flagging them as "missing mac_yield" or "BD slot is redundant" is signal noise, not a logic failure.
--- Other checks (transfer_hazards, gpu_portstore_shape, abi_handoff, enum_alias_membership, …) still apply to debug_skip declarations because real hazards / typos can still surface in them.
---
--- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table: --- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table:
--- `["static-analysis"] = { --- `["static-analysis"] = {
--- module = "passes.static_analysis", --- module = "passes.static_analysis",
--- kind = "diagnostic", --- kind = "diagnostic",
--- deps = {"word-counts", "components"}, --- deps = {"word-counts", "components"},
--- out = { { kind = "report", path_template = "<out_root>/<basename>.static_analysis.txt" } }
--- } --- }
--- `kind = "diagnostic"` keeps every finding visible in the report; the orchestrator does not exit non-zero on static-analysis errors. --- `kind = "diagnostic"` keeps every finding visible in the projection; the orchestrator does not exit non-zero on static-analysis errors.
--- Annotation and header-output validation remain build-stopping. --- Annotation and header-output validation remain build-stopping.
---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
@@ -148,6 +150,38 @@ local OUTPUT_EXTENSION = ".static_analysis.txt"
--- @field findings Finding[] -- findings for this atom --- @field findings Finding[] -- findings for this atom
--- @field total_cycles integer -- sum of token cycle costs --- @field total_cycles integer -- sum of token cycle costs
-- ════════════════════════════════════════════════════════════════════════════
-- Per-word-event helpers
-- ════════════════════════════════════════════════════════════════════════════
-- Pick the source-line field that best represents "where in the user's source file is this word?".
--
-- `word_events` (populated by `passes/emission_model.lua::stamp_root_provenance`) carry four line fields:
-- * `call_line` — physical line in the ROOT atom's source (the line of the `mac_X(...)` call site that triggered this emission, or `body_line` for direct words in the atom body)
-- * `body_line` — physical line in the body containing the emitted word (the atom body for direct words; the component body for words expanded inside `mac_X(...)`)
-- * `def_line` — line of the COMPONENT's declaration in its source file (only meaningful for words emitted inside a component expansion)
-- * `line` — body-relative line in the source text (not a physical source line; rarely useful in rendered findings)
--
-- For component-expanded words (e.g. the BD-slot nop of `jump_reg(R_AtomJmp)` inside `mac_yield()`),
-- `body_line` points into the COMPONENT's source file (e.g. `lottes_tape.h:110` for `ac_yield`'s body).
-- The user editing their atom body expects the line to point at THEIR source — i.e. the line where `mac_yield()`
-- was called (e.g. `hello_gte_tape.c:35`). That line is `call_line`.
--
-- For direct words in the atom body (no invocation wrapping them), `call_line == body_line` already,
-- so `call_line` works for both cases.
local function line_for_word_event(ev)
if ev == nil then return 0 end
return ev.call_line or ev.body_line or ev.line or ev.def_line or 0
end
-- True iff the given atom/component declaration has the bare `atom_dbg_skip` marker.
-- Used by the structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`,
-- `control_transfer_delay_slot_use`) to exempt runtime-helper declarations (`tape_exit`, `ac_yield`,
-- and the `ac_*` macro components) from findings whose contract they intentionally don't satisfy.
local function is_runtime_helper(atom)
return atom and atom.debug_skip == true
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- classify_tokens — per-token classification -- classify_tokens — per-token classification
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -551,11 +585,12 @@ local function append_cu2_finding(atom, event, forward, transition,
local event_ident = event.encoder or event.ident or "?" local event_ident = event.encoder or event.ident or "?"
local policy = duffle.CU2_TRANSITION_POLICY or {} local policy = duffle.CU2_TRANSITION_POLICY or {}
local evidence = policy.evidence or {} local evidence = policy.evidence or {}
local event_line = line_for_word_event(event)
atom.paths.hazards[#atom.paths.hazards + 1] = { atom.paths.hazards[#atom.paths.hazards + 1] = {
check = "transfer_hazards", check = "transfer_hazards",
kind = kind, kind = kind,
atom = atom.name, atom = atom.name,
line = event.body_line or event.line or event.def_line or 0, line = event_line,
source = event.def_path or event.source or "", source = event.def_path or event.source or "",
relation_id = "mtc0_cu2_visibility", relation_id = "mtc0_cu2_visibility",
semantic = "MTC0", semantic = "MTC0",
@@ -619,10 +654,11 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
local gap = ev_word - transition.producer_word - 1 local gap = ev_word - transition.producer_word - 1
local target = transition.target_state local target = transition.target_state
local event_line = line_for_word_event(event)
if target == "unknown" then if target == "unknown" then
append_cu2_finding(atom, event, forward, transition, gap, "info", "unknown", append_cu2_finding(atom, event, forward, transition, gap, "info", "unknown",
string.format("%s at line %d uses COP2 after an MTC0 Status write whose CU2 value is unknown (gap=%d, configured boundary=%d)" string.format("%s at line %d uses COP2 after an MTC0 Status write whose CU2 value is unknown (gap=%d, configured boundary=%d)"
, atom.name, event.body_line or event.line or event.def_line or 0 , atom.name, event_line
, gap, transition.required , gap, transition.required
) )
) )
@@ -635,7 +671,7 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
local verb = target == "enabled" and "enable" or "disable" local verb = target == "enabled" and "enable" or "disable"
append_cu2_finding(atom, event, forward, transition, gap, "warning", "conservative", append_cu2_finding(atom, event, forward, transition, gap, "warning", "conservative",
string.format("%s at line %d uses COP2 before the SR.CU2 %s transition has settled (gap=%d, required=%d; timing is conservative)" string.format("%s at line %d uses COP2 before the SR.CU2 %s transition has settled (gap=%d, required=%d; timing is conservative)"
, atom.name, event.body_line or event.line or event.def_line or 0 , atom.name, event_line
, verb, gap, transition.required , verb, gap, transition.required
) )
) )
@@ -653,7 +689,7 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
string.format( string.format(
"%s at line %d: COP2 unavailable after SR.CU2 was disabled" "%s at line %d: COP2 unavailable after SR.CU2 was disabled"
.. " (gap=%d, required=%d)", .. " (gap=%d, required=%d)",
atom.name, event.body_line or event.line or event.def_line or 0, atom.name, event_line,
gap, transition.required)) gap, transition.required))
forward.cu2_state = "disabled" forward.cu2_state = "disabled"
end end
@@ -708,7 +744,7 @@ local function analyze_hardware_relations(atom)
for _, ev in ipairs(events) do for _, ev in ipairs(events) do
local ev_ident = ev.encoder or ev.ident or "?" local ev_ident = ev.encoder or ev.ident or "?"
local ev_line = ev.body_line or ev.line or ev.def_line or 0 local ev_line = line_for_word_event(ev)
local ev_source = ev.def_path or ev.source or "" local ev_source = ev.def_path or ev.source or ""
local ev_args = ev.args or {} local ev_args = ev.args or {}
-- `word_events` use `i` as the 0-based word index across the entire expansion. -- `word_events` use `i` as the 0-based word index across the entire expansion.
@@ -867,7 +903,7 @@ local function analyze_hardware_relations(atom)
-- ── 4. Update semantic role state and stage post-command latch relations. ── -- ── 4. Update semantic role state and stage post-command latch relations. ──
-- A GTE command emits outputs with semantic roles (latest_screen_xy, otz, latest_color, etc.) per `duffle.GTE_COMMAND_OUTPUTS`. -- A GTE command emits outputs with semantic roles (latest_screen_xy, otz, latest_color, etc.) per `duffle.GTE_COMMAND_OUTPUTS`.
-- The walker records these on `forward_state.post_command_roles[<register>]` so the `gte_result_position` reader can later detect a reader that picks the wrong register. -- The walker records these on `forward_state.post_command_roles[<register>]` so the `gte_role_mismatch` reader can later detect a reader that picks the wrong register.
-- --
-- The walker also stages POST-COMMAND LATCH relations (kind = "command_latch_input"): a subsequent MTC2/CTC2 overwrite of a latched output before the measured boundary is a hazard. -- The walker also stages POST-COMMAND LATCH relations (kind = "command_latch_input"): a subsequent MTC2/CTC2 overwrite of a latched output before the measured boundary is a hazard.
-- The relation kind is intentionally separate from the preceding MTC2 → command relation (`MTC2` / `CTC2` / `LWC2`). -- The relation kind is intentionally separate from the preceding MTC2 → command relation (`MTC2` / `CTC2` / `LWC2`).
@@ -980,7 +1016,7 @@ local function check_gte_input_latch(atom, _pipe_ctx, findings)
end end
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
-- Check #1e: gte_result_position (READER for forward_state semantic roles). -- Check #1e: gte_role_mismatch (READER for forward_state semantic roles).
-- --
-- A GTE command emits outputs with semantic roles (latest_screen_xy, otz, latest_color, etc.) per `duffle.GTE_COMMAND_OUTPUTS`. -- A GTE command emits outputs with semantic roles (latest_screen_xy, otz, latest_color, etc.) per `duffle.GTE_COMMAND_OUTPUTS`.
-- The forward walker records `forward_state.post_command_roles[<register>]` after each command. -- The forward walker records `forward_state.post_command_roles[<register>]` after each command.
@@ -988,43 +1024,16 @@ end
-- A subsequent MFC2 (or any encoder that reads a C2 register) that picks the WRONG register for the active role emits a `result_role_mismatch` warning. -- A subsequent MFC2 (or any encoder that reads a C2 register) that picks the WRONG register for the active role emits a `result_role_mismatch` warning.
-- For example, reading `C2_SXY0` after RTPS is wrong: the `latest_screen_xy` role is `C2_SXY2`. -- For example, reading `C2_SXY0` after RTPS is wrong: the `latest_screen_xy` role is `C2_SXY2`.
-- --
-- Note: the OLD `gte_result_position` check also emitted table-gap info findings for `_post_<cmd>` components missing a row in `duffle.GTE_COMPONENT_RESULT_CONTRACTS`. That table-gap check was based on the `_post_<cmd>` NAMING convention rather than hardware truth, and was removed (the user did not want naming to encode ordering semantics; a proper `atom_info` directive for ordering semantics is a future TODO).
--
-- The first `transfer_hazards` reader comment above records the projection contract. -- The first `transfer_hazards` reader comment above records the projection contract.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
local function check_gte_result_position(atom, _pipe_ctx, findings) local function check_gte_role_mismatch(atom, _pipe_ctx, findings)
local forward = atom.paths and atom.paths.forward_state local forward = atom.paths and atom.paths.forward_state
if not forward or not forward.post_command_roles then return end if not forward or not forward.post_command_roles then return end
local events = atom.paths.word_events or {} local events = atom.paths.word_events or {}
-- Build a set of known _post_<cmd> component names whose contract rows we have to verify
-- (table-gap detection: a missing row key is itself an info finding).
-- The names are the BODY-LEVEL component calls that appear in atom body text;
-- The walker doesn't expose body tokens to the reader, so we scan the events' root_call_text.
local contracts = duffle.GTE_COMPONENT_RESULT_CONTRACTS or {}
local component_names_seen = {}
for _, ev in ipairs(events) do
local root_call = ev.root_call_text or ev.call_text or ""
local name = root_call:match("^([%w_]+)") or ""
if name:find("_post_") then component_names_seen[name] = true end
end
for component_name in pairs(component_names_seen) do
-- Strip any trailing parenthesized argument list / whitespace.
local bare = component_name:match("^([%w_]+)") or component_name
if contracts[bare] == nil then
findings[#findings + 1] = {
check = "gte_result_position",
kind = "info",
atom = atom.name,
line = 0,
source = "",
relation_id = "table_gap",
component_name = bare,
msg = string.format("%s: component %q has no GTE_COMPONENT_RESULT_CONTRACTS row (unknown _post_<cmd> contract)"
, atom.name, bare),
}
end
end
-- For each word event whose encoder is `gte_mv_from_data_r`, look up the register being read in `forward_state.post_command_roles`. -- For each word event whose encoder is `gte_mv_from_data_r`, look up the register being read in `forward_state.post_command_roles`.
-- If a role is set, the reader's register must match the role's register (the registered "latest_<role>" target). -- If a role is set, the reader's register must match the role's register (the registered "latest_<role>" target).
for _, ev in ipairs(events) do for _, ev in ipairs(events) do
@@ -1047,11 +1056,12 @@ local function check_gte_result_position(atom, _pipe_ctx, findings)
-- This is a semantic mismatch. -- This is a semantic mismatch.
if reg ~= latest_screen_xy_entry.command_register if reg ~= latest_screen_xy_entry.command_register
and (reg == "C2_SXY0" or reg == "C2_SXY1") then and (reg == "C2_SXY0" or reg == "C2_SXY1") then
local ev_line = line_for_word_event(ev)
findings[#findings + 1] = { findings[#findings + 1] = {
check = "gte_result_position", check = "gte_role_mismatch",
kind = "warning", kind = "warning",
atom = atom.name, atom = atom.name,
line = ev.body_line or ev.line or ev.def_line or 0, line = ev_line,
source = ev.def_path or ev.source or "", source = ev.def_path or ev.source or "",
relation_id = "result_role_mismatch", relation_id = "result_role_mismatch",
semantic = "result_position", semantic = "result_position",
@@ -1062,7 +1072,7 @@ local function check_gte_result_position(atom, _pipe_ctx, findings)
producer_word = latest_screen_xy_entry.producer_word, producer_word = latest_screen_xy_entry.producer_word,
producer_line = latest_screen_xy_entry.producer_line, producer_line = latest_screen_xy_entry.producer_line,
msg = string.format("%s at line %d: reading %s after %s but the %s role is C2_SXY2 (not %s)" msg = string.format("%s at line %d: reading %s after %s but the %s role is C2_SXY2 (not %s)"
, atom.name, ev.body_line or ev.line or ev.def_line or 0 , atom.name, ev_line
, reg, latest_screen_xy_entry.command , reg, latest_screen_xy_entry.command
, latest_screen_xy_entry.role , latest_screen_xy_entry.role
, reg), , reg),
@@ -1084,6 +1094,13 @@ end
-- Branch/jump delay-slot NOPs belong to `control_transfer_delay_slot_use`, so this check leaves them unclassified. -- Branch/jump delay-slot NOPs belong to `control_transfer_delay_slot_use`, so this check leaves them unclassified.
-- The fixed `mac_yield()` handshake (`jump_reg(R_AtomJmp), nop`) is preserved as suppressed. -- The fixed `mac_yield()` handshake (`jump_reg(R_AtomJmp), nop`) is preserved as suppressed.
-- --
-- Both classifications emit at `info` severity: `modeled-required` documents the model boundary and `modeled-redundant`
-- is a soft observation ("you have a redundant nop; consider replacing it").
-- Neither is a logic failure, so neither rises to `warning`.
--
-- `atom_dbg_skip` runtime helpers (`tape_exit`, `ac_yield`, the `ac_*` macro components) are exempt:
-- their structural nops are part of the fixed handshake and not author choices.
--
-- The first `transfer_hazards` reader comment above records the projection contract. -- The first `transfer_hazards` reader comment above records the projection contract.
-- ───────────────────────────────────────────────────────────────────────── -- ─────────────────────────────────────────────────────────────────────────
@@ -1091,6 +1108,9 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
local forward = atom.paths and atom.paths.forward_state local forward = atom.paths and atom.paths.forward_state
local events = atom.paths.word_events or {} local events = atom.paths.word_events or {}
if not events or #events == 0 then return end if not events or #events == 0 then return end
-- Runtime-helper atoms / components (e.g. tape_exit, ac_yield) carry `debug_skip = true` from the bare
-- `atom_dbg_skip` marker; their structural nops are part of the fixed handshake and not author choices.
if is_runtime_helper(atom) then return end
-- The walker does not currently snapshot the pending state per event; we replay the same forward walk cheaply here. -- The walker does not currently snapshot the pending state per event; we replay the same forward walk cheaply here.
-- The replay is observation-only (no staging); the only output is one finding per non-BD-slot nop with its classification. -- The replay is observation-only (no staging); the only output is one finding per non-BD-slot nop with its classification.
@@ -1100,20 +1120,16 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
local ev_ident = ev.encoder or "" local ev_ident = ev.encoder or ""
local ev_args = ev.args or {} local ev_args = ev.args or {}
local ev_word = ev.i or 0 local ev_word = ev.i or 0
local ev_line = line_for_word_event(ev)
-- Classify the nop BEFORE its event is applied to the pending state. -- Classify the nop BEFORE its event is applied to the pending state.
if ev_ident == "nop" and prev_ev ~= nil then if ev_ident == "nop" and prev_ev ~= nil then
-- Skip BD-slot nops: they are exclusively owned by control_transfer_delay_slot_use. -- Skip BD-slot nops: they are exclusively owned by control_transfer_delay_slot_use.
local prev_ident = prev_ev.encoder or "" -- Every BD-slot nop is structural; this check never reports on it.
local prev_args = prev_ev.args or {} -- (The earlier `if not suppressed then is_bd_slot = true end` form inverted the suppression — the `mac_yield()` handshake's `jump_reg(R_AtomJmp)` was incorrectly flagged.)
local bd_policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {} local prev_ident = prev_ev.encoder or ""
local is_bd_slot = false local bd_policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {}
local policy = bd_policies[prev_ident] local is_bd_slot = bd_policies[prev_ident] ~= nil
if policy then
local arg1 = prev_args[1]
local suppressed = policy.suppress_arg1 and policy.suppress_arg1[arg1] or nil
if not suppressed then is_bd_slot = true end
end
if not is_bd_slot then if not is_bd_slot then
-- Find a pending modeled relation that this nop would retire. -- Find a pending modeled relation that this nop would retire.
local retired = nil local retired = nil
@@ -1157,7 +1173,7 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
check = "hazard_nop_use", check = "hazard_nop_use",
kind = "info", kind = "info",
atom = atom.name, atom = atom.name,
line = ev.body_line or ev.line or ev.def_line or 0, line = ev_line,
source = ev.def_path or ev.source or "", source = ev.def_path or ev.source or "",
nop_classification = "modeled-required", nop_classification = "modeled-required",
nop_word_index = ev_word, nop_word_index = ev_word,
@@ -1165,7 +1181,7 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
producer_destination = retired.destination, producer_destination = retired.destination,
consumer_token = would_be_consumer or "<would-be-consumer>", consumer_token = would_be_consumer or "<would-be-consumer>",
msg = string.format("%s at line %d: nop at word %d is modeled-required (retires %s for %s)" msg = string.format("%s at line %d: nop at word %d is modeled-required (retires %s for %s)"
, atom.name, ev.body_line or ev.line or ev.def_line or 0, ev_word, retired.relation.id, retired.destination , atom.name, ev_line, ev_word, retired.relation.id, retired.destination
), ),
} }
else else
@@ -1173,16 +1189,16 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
local slot_kind = "plain" local slot_kind = "plain"
findings[#findings + 1] = { findings[#findings + 1] = {
check = "hazard_nop_use", check = "hazard_nop_use",
kind = "warning", kind = "info",
atom = atom.name, atom = atom.name,
line = ev.body_line or ev.line or ev.def_line or 0, line = ev_line,
source = ev.def_path or ev.source or "", source = ev.def_path or ev.source or "",
nop_classification = "modeled-redundant", nop_classification = "modeled-redundant",
nop_word_index = ev_word, nop_word_index = ev_word,
retired_relation = nil, retired_relation = nil,
slot_kind = slot_kind, slot_kind = slot_kind,
msg = string.format("%s at line %d: nop at word %d is modeled-redundant (no pending modeled relation)" msg = string.format("%s at line %d: nop at word %d is modeled-redundant (no pending modeled relation)"
, atom.name, ev.body_line or ev.line or ev.def_line or 0, ev_word , atom.name, ev_line, ev_word
), ),
} }
end end
@@ -1254,6 +1270,9 @@ end
-- Suppress the finding when `policy.suppress_arg1[first_arg]` is non-nil. -- Suppress the finding when `policy.suppress_arg1[first_arg]` is non-nil.
-- The only current suppression is `jump_reg(R_AtomJmp)`, the fixed `mac_yield()` handshake. -- The only current suppression is `jump_reg(R_AtomJmp)`, the fixed `mac_yield()` handshake.
-- --
-- `atom_dbg_skip` runtime helpers (`tape_exit`, `ac_yield`, the `ac_*` macro components) are exempt:
-- their BD slots are part of the fixed handshake (`jump_reg(rret_addr), nop` for tape_exit, `jump_reg(R_AtomJmp), nop` for ac_yield).
--
-- `pipe_ctx` is unused; the uniform `(atom, pipe_ctx, findings)` signature is preserved so the check plugs into -- `pipe_ctx` is unused; the uniform `(atom, pipe_ctx, findings)` signature is preserved so the check plugs into
-- the existing CHECK_RULES dispatch without modifying the per-atom loop or analyze_atom_paths. -- the existing CHECK_RULES dispatch without modifying the per-atom loop or analyze_atom_paths.
-- `passes/emission_model` already normalizes `nop2` to two `nop` events and `atom_label` to zero events, so no special-case branching is needed for either. -- `passes/emission_model` already normalizes `nop2` to two `nop` events and `atom_label` to zero events, so no special-case branching is needed for either.
@@ -1262,6 +1281,9 @@ end
local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings) local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings)
local events = atom.paths.word_events or {} local events = atom.paths.word_events or {}
if not events or #events == 0 then return end if not events or #events == 0 then return end
-- Runtime-helper atoms / components (e.g. tape_exit, ac_yield) carry `debug_skip = true` from the bare
-- `atom_dbg_skip` marker; their structural BD slots are part of the fixed handshake.
if is_runtime_helper(atom) then return end
local policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {} local policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {}
for event_idx, event in ipairs(events) do for event_idx, event in ipairs(events) do
-- Canonical word_events use `encoder` as the leading identifier of the emitting token). -- Canonical word_events use `encoder` as the leading identifier of the emitting token).
@@ -1276,9 +1298,9 @@ local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings)
local slot = events[event_idx + 1] local slot = events[event_idx + 1]
local slot_ident = slot and (slot.encoder or slot.ident) or "<missing>" local slot_ident = slot and (slot.encoder or slot.ident) or "<missing>"
if slot == nil or (slot.encoder or slot.ident) == "nop" then if slot == nil or (slot.encoder or slot.ident) == "nop" then
-- Each word event carries `body_line` as the physical source line. -- Prefer `call_line` (the line of the `mac_X(...)` call site in the atom body) so the rendered
-- Use `body_line`, then `def_line`, then 0. -- finding points at the user's source, not at the vendored component body.
local ev_line = event.body_line or event.line or event.def_line or 0 local ev_line = line_for_word_event(event)
findings[#findings + 1] = { findings[#findings + 1] = {
atom = atom.name, atom = atom.name,
line = ev_line, line = ev_line,
@@ -1303,8 +1325,17 @@ end
--- Empty bodies are not currently flagged — runtime infrastructure atoms like --- Empty bodies are not currently flagged — runtime infrastructure atoms like
--- `MipsAtom_(yield) { mac_yield() }` and `MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }` --- `MipsAtom_(yield) { mac_yield() }` and `MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }`
--- are valid as-is; mac_yield at the end is the contract. --- are valid as-is; mac_yield at the end is the contract.
---
--- Runtime helpers carrying the bare `atom_dbg_skip` marker (`tape_exit`, `ac_yield`, the `ac_*` macro components) are exempt:
--- they intentionally do not follow the standard "1 yield at the end" contract. `tape_exit` performs its own `jump_reg(rret_addr),
--- nop` to return from the tape runner; `ac_yield` IS the `mac_yield()` implementation.
--- Flagging them as "missing mac_yield" is signal noise, not a logic failure.
---
--- Uses the standard `(atom, pipe_ctx, findings)` signature; `pipe_ctx` is unused. --- Uses the standard `(atom, pipe_ctx, findings)` signature; `pipe_ctx` is unused.
local function check_mac_yield_uniformity(atom, pipe_ctx, findings) local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
-- Runtime-helper atoms / components (e.g. tape_exit, ac_yield) carry `debug_skip = true` from the bare
-- `atom_dbg_skip` marker; they intentionally break the "1 yield at the end" contract.
if is_runtime_helper(atom) then return end
-- Per-kind semantics: -- Per-kind semantics:
-- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job. -- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job.
-- MipsAtomComp_ (bare static-array component): ZERO mac_yield. -- MipsAtomComp_ (bare static-array component): ZERO mac_yield.
@@ -1935,7 +1966,7 @@ end
local CHECK_RULES = { local CHECK_RULES = {
{ name = "transfer_hazards", per_atom = check_transfer_hazards }, { name = "transfer_hazards", per_atom = check_transfer_hazards },
{ name = "gte_input_latch", per_atom = check_gte_input_latch }, { name = "gte_input_latch", per_atom = check_gte_input_latch },
{ name = "gte_result_position", per_atom = check_gte_result_position }, { name = "gte_role_mismatch", per_atom = check_gte_role_mismatch },
{ name = "hazard_nop_use", per_atom = check_hazard_nop_use }, { name = "hazard_nop_use", per_atom = check_hazard_nop_use },
{ name = "control_transfer_delay_slot_use",per_atom = check_control_transfer_delay_slot_use}, { name = "control_transfer_delay_slot_use",per_atom = check_control_transfer_delay_slot_use},
{ name = "mac_yield_uniformity", per_atom = check_mac_yield_uniformity }, { name = "mac_yield_uniformity", per_atom = check_mac_yield_uniformity },
@@ -2179,203 +2210,6 @@ local function validate(ctx, src, corpus_pipe_ctx)
} }
end end
-- ════════════════════════════════════════════════════════════════════════════
-- Per-directory output: build/gen/<dir_basename>.static_analysis.txt
-- ════════════════════════════════════════════════════════════════════════════
--- Per-directory emit. Aggregates atoms + findings across every source in `dir_sources`
--- and writes a single report to `<out_root>/<dir_basename>.static_analysis.txt`.
--- Called only when at least one atom was found (the caller in M.run handles the skip).
---
--- `info` is finding-level info only (kind == "info" findings); the scanned/cycles summary rows
--- live in `summaries` and are rendered as trailing summary lines after `Module findings:`.
local function emit_module_static_analysis_txt(ctx, dir, dir_sources, atoms, findings, errors, warnings, info, summaries)
-- Module basename = last component of `dir` ("code/duffle" -> "duffle").
local dir_basename = dir:match("([^/\\]+)$") or dir
local out_path = ctx.out_root .. "/" .. dir_basename .. ".static_analysis.txt"
duffle.ensure_dir(ctx.out_root)
local lines = {}
local function add(s) lines[#lines + 1] = s end
add("========================================================")
add("STATIC ANALYSIS PASS -- module " .. dir_basename)
add("========================================================")
add(string.format("Sources: %d", #dir_sources))
for _, s in ipairs(dir_sources) do
add(" " .. s.path)
end
add("")
-- Tally atoms by kind for the header summary
local n_atoms, n_bare, n_proc = 0, 0, 0
for _, a in ipairs(atoms) do
n_atoms = n_atoms + 1
if a.kind == "comp_bare" then n_bare = n_bare + 1
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
end
end
local header_atoms = string.format("Atoms: %d", n_atoms)
if n_bare > 0 or n_proc > 0 then
header_atoms = header_atoms .. string.format(" (atoms: %d, comp_bare: %d, comp_proc: %d)",
n_atoms - n_bare - n_proc, n_bare, n_proc)
end
-- Header carries the per-severity counts; info is its own column, not a warning.
-- (`Info: N` is the byte-asserted field that the focused test matches; do not collapse it into Warnings.)
add(string.format("%s Findings: %d Errors: %d Warnings: %d Info: %d",
header_atoms, #findings, #errors, #warnings, #info))
add("")
-- Group findings by atom (with source prefix when multi-source module)
local multi_source = #dir_sources > 1
local by_atom = {}
for _, f in ipairs(findings) do
by_atom[f.atom] = by_atom[f.atom] or {}
by_atom[f.atom][#by_atom[f.atom] + 1] = f
end
if next(by_atom) == nil then
add(" (no findings -- every atom passed all checks)")
else
add("── Findings by atom ─────────────────────────────────────")
for _, a in ipairs(atoms) do
local fs = by_atom[a.name]
if fs then
local label = a.name
if multi_source and a.source_path then
label = string.format("%s (%s)", a.name, a.source_path:match("([^/\\]+)$") or a.source_path)
end
add(string.format(" %s line %d", label, a.line))
for _, f in ipairs(fs) do
add(string.format(" [%s] %s", f.check, f.msg))
end
end
end
end
add("")
add("── Errors ──────────────────────────────────────────────")
if #errors == 0 then add(" (none)") end
for _, e in ipairs(errors) do
add(string.format(" X line %d %s", e.line, e.msg))
end
add("")
add("── Warnings ────────────────────────────────────────────")
if #warnings == 0 then add(" (none)") end
for _, w in ipairs(warnings) do
add(string.format(" ! line %d %s", w.line, w.msg))
end
-- Finding-level Info section.
-- Rendered between Warnings and the per-atom cycle table so the next `── ` line after `── Info` is the per-atom cycle counts section;
-- the trailing scan/cycle summary rows (rendered after Module findings) stay outside this section.
add("")
add("── Info ────────────────────────────────────────────────")
if #info == 0 then add(" (none)") end
for _, i_ in ipairs(info) do
add(string.format(" i line %d %s", i_.line, i_.msg))
end
-- Per-atom cycle counts (path-aware). For each atom:
-- min = shortest path through the body (earliest exit)
-- max = longest path through the body (full fall-through)
-- br = number of branch instructions
-- paths = number of distinct paths reached
-- Both min and max are best-case (no stalls); BD-slot nops are absorbed into branch costs (MIPS semantics).
add("")
add("── Per-atom cycle counts (path-aware, best case, no stalls) ─")
if #atoms == 0 then
add(" (no atoms)")
else
-- Sort atoms by max cycles descending for quick scanning.
local sorted = {}
for _, a in ipairs(atoms) do sorted[#sorted + 1] = a end
table.sort(sorted, function(x, y) return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0) end)
for _, a in ipairs(sorted) do
local p = a.paths or {}
local br_count = p.branches or 0
local path_count = p.paths or 0
local loops_tag = p.has_loops and " [loop!]" or ""
local unknown_tag = ""
if p.unknown_macros and #p.unknown_macros > 0 then
unknown_tag = string.format(" [unknown: %s]",
table.concat(p.unknown_macros, ", "))
end
local name_label = a.name
if multi_source and a.source_path then
name_label = string.format("%s (%s)", a.name, a.source_path:match("([^/\\]+)$") or a.source_path)
end
if br_count > 0 then
add(string.format(" %-44s min=%4d max=%4d br=%d paths=%d (line %d)%s%s",
name_label, p.cycles_min or 0, p.cycles_max or 0, br_count, path_count,
a.line, loops_tag, unknown_tag))
else
add(string.format(" %-44s %4d cycles (line %d, no branches)%s%s",
name_label, p.cycles_min or 0, a.line, loops_tag, unknown_tag))
end
end
end
add("")
add("── Per-source scan summary ──────────────────────────────")
-- One line per source that contributed atoms.
-- The line includes the source basename + per-source atom count + (if path-aware cycle data is present) the min..max cycle range.
-- Sources with 0 atoms are skipped (they're just header files that declared no MipsAtom_ — they're already listed in the module's "Sources:" section above).
for _, src in ipairs(dir_sources) do
local src_atoms = {}
for _, a in ipairs(atoms) do
if a.source_path == src.path then
src_atoms[#src_atoms + 1] = a
end
end
if #src_atoms == 0 then
goto continue
end
local atom_count = #src_atoms
local mn, mx = math.huge, -1
for _, a in ipairs(src_atoms) do
local p = a.paths or {}
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
end
local path_str
if mx > 0 then
path_str = string.format(" cycles=%d..%d", mn, mx)
else
path_str = string.format(" %d cycles", mn)
end
add(string.format(" %-30s %d atom%s%s",
src.basename, atom_count,
atom_count == 1 and "" or "s",
path_str))
::continue::
end
-- Module-level findings summary (across all sources).
-- Info has its own count; it remains separate from warnings.
local total_errs = #errors
local total_warns = #warnings
local total_infos = #info
add("")
add(string.format("Module findings: %d error(s), %d warning(s), %d info", total_errs, total_warns, total_infos))
-- Per-source "scanned:" / "cycles:" summary lines (each line includes the source basename for traceability).
-- These are kept SEPARATE from the finding-level Info section above so the report's Info section is signal-only
-- (true findings), not a mix of findings + rollups.
-- The downstream test (`test_control_transfer_delay_slot.lua`)
-- asserts that the Info section contains NEITHER `scanned:` NOR `cycles:` lines.
if summaries and #summaries > 0 then
add("")
for _, s in ipairs(summaries) do
add(string.format(" %s", s.msg))
end
end
duffle.write_file(out_path, table.concat(lines, "\n") .. "\n")
return out_path
end
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- M.run — orchestrator entry -- M.run — orchestrator entry
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -2431,21 +2265,30 @@ function M.run(ctx)
for _, s in ipairs(result.summaries or {}) do dir_summaries[#dir_summaries + 1] = s end for _, s in ipairs(result.summaries or {}) do dir_summaries[#dir_summaries + 1] = s end
end end
-- Skip directories with zero atoms. A directory with only headers / no MipsAtom_ is "nothing to report". -- Stash per-module results on the corpus for `report.lua` to consume.
if #all_atoms == 0 then -- Avoids re-running validate() in the report pass + avoids rebuilding corpus_pipe_ctx.
-- Still aggregate errors/warnings/info so orchestrator sees them, but don't write a file. -- Pattern matches `corpus.atoms_by_name` / `corpus.word_counts` / `corpus.components`
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end -- (one writer: `static_analysis.lua`; one reader: `report.lua`).
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end -- Module basename = last component of `dir` ("code/duffle" -> "duffle").
for _, i_ in ipairs(dir_info) do info[#info + 1] = i_ end local dir_basename = dir:match("([^/\\]+)$") or dir
else corpus.static_analysis_results = corpus.static_analysis_results or {}
local out_path = emit_module_static_analysis_txt(ctx, dir, dir_sources, all_atoms, all_findings, dir_errors, dir_warnings, dir_info, dir_summaries) corpus.static_analysis_results[dir_basename] = {
if out_path then atoms = all_atoms,
table.insert(outputs, { static_analysis_txt = out_path }) findings = all_findings,
end errors = dir_errors,
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end warnings = dir_warnings,
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end info = dir_info,
for _, i_ in ipairs(dir_info) do info[#info + 1] = i_ end summaries = dir_summaries,
end sources = dir_sources,
}
-- Aggregate per-dir errors/warnings/info into the orchestrator totals.
-- Hoisted out of any per-dir file-emit so `report.lua` can drop the on-disk file emitter without losing the cross-module rollup.
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end
for _, i_ in ipairs(dir_info) do info [#info + 1] = i_ end
-- (No per-dir emit: per-module findings are stashed on `corpus.static_analysis_results` above.
-- `report.lua` reads that projection to render `<module>.atom_meta_report.md` without re-running validate().)
end end
-- Result exposes at least {outputs, errors, warnings, info}. -- Result exposes at least {outputs, errors, warnings, info}.
+7 -9
View File
@@ -2,17 +2,14 @@
--- ---
--- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection). --- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection).
--- ---
--- **Architecture**: --- Architecture:
--- - **PASSES table** — declarative dep graph (data, not code). --- - PASSES table: Declarative dep graph (data, not code).
--- - **FLAG_HANDLERS table** — maps CLI flags to handlers. --- - FLAG_HANDLERS table: Maps CLI flags to handlers.
--- - **parse_args****build_ctx** (resolves unity/direct includes or exact sources; no semantic scanning) → **topo_sort****dispatch_passes**. --- - parse_args → build_ctx (resolves unity/direct includes or exact sources) → topo_sort → dispatch_passes.
--- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`). --- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`).
--- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`. --- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`.
--- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only. --- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only.
--- ---
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
--- Lua 5.3 compatible.
---
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup -- Module-scope requires + package.path setup
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -160,7 +157,7 @@ local PASSES = {
report = { report = {
module = "passes.report", module = "passes.report",
kind = "report", kind = "report",
deps = {"annotation", "static-analysis"}, deps = {"annotation", "static-analysis", "atoms-source-map"}, -- +atoms-source-map (consolidated-report-files refactor, 2026-07-26)
groups = { "pre-link" }, groups = { "pre-link" },
}, },
} }
@@ -206,7 +203,8 @@ end
-- Pass-kind taxonomy: Which kinds stop the build on errors? -- Pass-kind taxonomy: Which kinds stop the build on errors?
-- --
-- Report severity is independent from process exit policy. A "diagnostic" pass still writes every `error`/`warning` finding into its report file, -- Report severity is independent from process exit policy.
-- A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
-- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero. -- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero.
-- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true". -- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true".
local PASS_KIND_STOP_ON_ERROR = { local PASS_KIND_STOP_ON_ERROR = {