mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-08-06 07:38:47 +00:00
Compare commits
16
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
54a5bb9a31 | ||
|
|
e0f4ac873d | ||
|
|
f17fa9165e | ||
|
|
8282f8e902 | ||
|
|
9eb696ece8 | ||
|
|
858e57f293 | ||
|
|
afcd9b86f0 | ||
|
|
43cd4e0344 | ||
|
|
09dde54030 | ||
|
|
315e1b2c5e | ||
|
|
02658d3609 | ||
|
|
dbc459b7e0 | ||
|
|
a704341fc6 | ||
|
|
7421b32fd7 | ||
|
|
e2eb74be19 | ||
|
|
338f1fe46e |
@@ -17,3 +17,6 @@ toolchain/PSn00bSDK
|
||||
.vscode/settings.json
|
||||
toolchain/lfs
|
||||
toolchain/lpeg
|
||||
|
||||
scratch
|
||||
toolchain/libpsn00b
|
||||
|
||||
Vendored
+36
-36
@@ -74,41 +74,7 @@
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello GTE Psy-Q!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}/build",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_gte.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_gte.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load hello_gte.elf",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello GTE Psy-Q! (atoms debug — DWARF-injected)",
|
||||
"name": "Debug: Hello GTE!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
@@ -138,7 +104,41 @@
|
||||
"monitor reset shellhalt",
|
||||
"load build/hello_gte.dwarf-injected.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"source build/gen/hello_gte.gdbinit",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "Debug: Hello Joypad!",
|
||||
"type": "gdb",
|
||||
"request": "attach",
|
||||
"target": "localhost:3333",
|
||||
"remote": true,
|
||||
"cwd": "${workspaceRoot}",
|
||||
"valuesFormatting": "parseText",
|
||||
"registerLimit": "1-32",
|
||||
"frameFilters": false,
|
||||
"showDevDebugOutput": false,
|
||||
"printCalls": false,
|
||||
"stopAtConnect": true,
|
||||
"gdbpath": "gdb-multiarch",
|
||||
"windows": {
|
||||
"gdbpath": "gdb-multiarch.exe"
|
||||
},
|
||||
"osx": {
|
||||
"gdbpath": "gdb"
|
||||
},
|
||||
"executable": "${workspaceRoot}/build/hello_joypad.dwarf-injected.elf",
|
||||
"setupCommands": [
|
||||
{ "text": "set mi-async off" },
|
||||
{ "text": "set remotetimeout 0" },
|
||||
{ "text": "set logging file build/gen/hello_joypad.gdb.log" },
|
||||
{ "text": "set logging redirect on" }
|
||||
],
|
||||
"autorun": [
|
||||
"monitor reset shellhalt",
|
||||
"load build/hello_joypad.dwarf-injected.elf",
|
||||
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||
"tbreak main",
|
||||
"continue"
|
||||
]
|
||||
|
||||
+11
-10
@@ -135,16 +135,17 @@ enum { false = 0, true = 1, true_overflow, };
|
||||
|
||||
typedef void Proc_(VoidFn) (void);
|
||||
|
||||
#define kilo(n) (C_(U4, n) << 10)
|
||||
#define mega(n) (C_(U4, n) << 20)
|
||||
#define giga(n) (C_(U4, n) << 30)
|
||||
#define tera(n) (C_(U4, n) << 40)
|
||||
#define null C_(U4, 0)
|
||||
#define nullptr C_(void*, 0)
|
||||
#define O_(type, field) (C_(U4, & C_(type*,0)->field))
|
||||
|
||||
#define OT_(field) O_(typeof_ptr(& field), filed))
|
||||
#define S_(data) C_(U4, sizeof(data))
|
||||
#define kilo(n) (C_(U4, n) << 10)
|
||||
#define mega(n) (C_(U4, n) << 20)
|
||||
#define giga(n) (C_(U4, n) << 30)
|
||||
#define tera(n) (C_(U4, n) << 40)
|
||||
|
||||
#define null C_(U4, 0)
|
||||
#define nullptr C_(void*, 0)
|
||||
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
||||
#define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
|
||||
#define OT_(field) O_(typeof_ptr(& field), filed))
|
||||
#define S_(data) C_(U4, sizeof(data))
|
||||
|
||||
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
||||
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
|
||||
|
||||
+15
-23
@@ -50,17 +50,13 @@
|
||||
#define asm_words(...) m_expand(glue(GCC_ASM_INL_, GCC_ASM_COUNT_ARGS(__VA_ARGS__))(__VA_ARGS__))
|
||||
// Very nasty macro expansion. See the Cruft pragma region after all the DSL defines
|
||||
|
||||
/* reg_str(n) — Stringify an integer register id into the GCC asm
|
||||
* string form (e.g. 12 → "$12"). Use this anywhere GCC's parser
|
||||
* expects a literal string identifying a register: clobber lists,
|
||||
* asm templates, etc. The two-level macro is the standard preprocessor
|
||||
* idiom for forcing one level of expansion before stringify — without
|
||||
* it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead
|
||||
* of expanding `R_T4` to its value first.
|
||||
/* reg_str(n) — Stringify an integer register id into the GCC asm string form (e.g. 12 → "$12").
|
||||
* Use this anywhere GCC's parser expects a literal string identifying a register: clobber lists,
|
||||
* asm templates, etc. The two-level macro is the standard preprocessor idiom for forcing one level of expansion before stringify —
|
||||
* without it, `#n` would stringify the macro name `R_T4` to `"R_T4"` instead of expanding `R_T4` to its value first.
|
||||
*
|
||||
* For declaring a register variable bound to a specific GPR, use the
|
||||
* `rgcc(n)` bundle from gcc_asm.h instead — it adds the `__asm__()`
|
||||
* qualifier around the string.
|
||||
* For declaring a register variable bound to a specific GPR, use the `rgcc(n)` bundle from gcc_asm.h instead —
|
||||
* it adds the `__asm__()` qualifier around the string.
|
||||
*
|
||||
* register V3_S2* p0 __asm__(reg_str(R_T4)) = ...; // verbose
|
||||
* register V3_S2* p0 rgcc(R_T4) = ...; // bundled
|
||||
@@ -85,21 +81,19 @@
|
||||
* - The string "$12" is derived from it via reg_str, so they cannot drift apart.
|
||||
* - Spelling `__asm__(reg_str(R_T4_Code))` at every call site is noise.
|
||||
*
|
||||
* tmpl defined in dsl.h (the token-paste glue).
|
||||
* tmpl defined in dsl.h (token-paste glue).
|
||||
* rgcc define here (gcc_asm.h) because the `__asm__` keyword is GCC-specific.
|
||||
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
||||
* Anyone porting to a different compiler's asm dialect overrides rgcc,
|
||||
* and the integer→string derivation in rlit can be retargeted in one place.
|
||||
*
|
||||
* For clobber lists and asm-template strings, use the bare `rlit(R_T4_Code)`.
|
||||
* ------------------------------------------------------------------------ */
|
||||
#define rgcc(n) __asm__(rlit(n))
|
||||
|
||||
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used
|
||||
* by the placeholder-pun macros (the .word bodies are fully baked
|
||||
* at compile time and have no runtime operand references), but kept
|
||||
* here for completeness in case a future asm template needs to refer
|
||||
* to a runtime input by position. Mirror of rgcc but produces "%N"
|
||||
* instead of "$N". */
|
||||
/* rgcc_ref(n) — GCC operand-reference form "%N". Not currently used by the placeholder-pun macros
|
||||
* (the .word bodies are fully baked at compile time and have no runtime operand references),
|
||||
* but kept here for completeness in case a future asm template needs to refer to a runtime input by position.
|
||||
* Mirror of rgcc but produces "%N" instead of "$N". */
|
||||
#define rgcc_ref_(n) "%" #n
|
||||
#define rgcc_ref(n) rgcc_ref_(n)
|
||||
|
||||
@@ -147,11 +141,9 @@
|
||||
9, 8, 7, 6, 5, 4, 3, 2, 1, 0))
|
||||
|
||||
/* --- 2. String Concatenation Helpers --- *
|
||||
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's
|
||||
* asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
||||
* The `%cN` form is for printing *character* constants; for arbitrary
|
||||
* integer immediates (the only kind `"i"(...)` produces), the plain
|
||||
* `%N` form is the right one. Both expand to the bare immediate.
|
||||
* NOTE: we use `%0`, `%1`, ... not `%c0`, `%c1`, ... because GCC's asm-parser rejects `%cN` in this position with "invalid use of '%c'".
|
||||
* The `%cN` form is for printing *character* constants; for arbitrary integer immediates (the only kind `"i"(...)` produces),
|
||||
* the plain `%N` form is the right one. Both expand to the bare immediate.
|
||||
*/
|
||||
#define GCC_ASM_W1 "%0"
|
||||
#define GCC_ASM_W2 GCC_ASM_W1 ", %1"
|
||||
|
||||
@@ -98,11 +98,11 @@ WORD_COUNT(mac_format_f3_color, 3)
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
#define mac_gte_store_f3_post_rtpt(...) \
|
||||
#define mac_gte_store_f3(...) \
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
|
||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
|
||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
|
||||
WORD_COUNT(mac_gte_store_f3_post_rtpt, 3)
|
||||
WORD_COUNT(mac_gte_store_f3, 3)
|
||||
|
||||
#define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||
@@ -115,25 +115,20 @@ WORD_COUNT(mac_format_g4_color, 12)
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
|
||||
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
|
||||
* three registers aligned with v0/v1/v2 you must store before RTPS).
|
||||
* The macro name declares the pipeline position; check #6 (GTE state-
|
||||
* machine validation) verifies the call site matches the declaration. */
|
||||
#define mac_gte_store_g4_p012_post_rtpt_pre_rtps(...) \
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||
#define mac_gte_store_g4_p012(...) \
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \
|
||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \
|
||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
|
||||
WORD_COUNT(mac_gte_store_g4_p012_post_rtpt_pre_rtps, 3)
|
||||
WORD_COUNT(mac_gte_store_g4_p012, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
|
||||
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
|
||||
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
|
||||
* prevents).
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||
*/
|
||||
#define mac_gte_store_g4_p3_post_rtps(...) \
|
||||
#define mac_gte_store_g4_p3(...) \
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3))
|
||||
WORD_COUNT(mac_gte_store_g4_p3_post_rtps, 1)
|
||||
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||
|
||||
|
||||
+291
-243
@@ -39,8 +39,8 @@
|
||||
/* ============================================================================
|
||||
* Hardware MMIO Addresses
|
||||
* ============================================================================
|
||||
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2
|
||||
* 0x1F800000+. GP0 (offset 0x10) is the data port (commands + params).
|
||||
* PSX GPU has two 32-bit ports in the I/O register region at KSEG2 0x1F800000+.
|
||||
* GP0 (offset 0x10) is the data port (commands + params).
|
||||
* GP1 (offset 0x14) is the control port (status, ctrl writes).
|
||||
* ============================================================================ */
|
||||
/* IO base address (KSEG2 0x1F800000+ for the I/O register region).
|
||||
@@ -49,18 +49,18 @@
|
||||
* `lui $reg, 0x1F80` (1 word) then `sw $data, GPIO_PORT*_OFFSET($reg)` (1 word).
|
||||
* Mirrors the `IO_BASE_ADDR equ 0x1F80` + `gpio_port0 equ 0x1810` pattern from graphics_hello/gp.s. */
|
||||
enum {
|
||||
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
|
||||
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
|
||||
IO_BASE_ADDR = 0x1F800000, /* full 32-bit I/O region base */
|
||||
IO_BASE_ADDR_HI16 = 0x1F80, /* fits in a single `lui $reg, 0x1F80` */
|
||||
|
||||
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
|
||||
* that pin a register to IO_BASE_ADDR and access ports via offsets:
|
||||
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
|
||||
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
|
||||
GPIO_PORT0_OFFSET = 0x1810,
|
||||
GPIO_PORT1_OFFSET = 0x1814,
|
||||
/* Offsets from IO_BASE_ADDR to each port. Used by tape-side macros
|
||||
* that pin a register to IO_BASE_ADDR and access ports via offsets:
|
||||
* sw $data, GPIO_PORT0_OFFSET($io_base) ; write GP0
|
||||
* sw $data, GPIO_PORT1_OFFSET($io_base) ; write GP1 */
|
||||
GPIO_PORT0_OFFSET = 0x1810,
|
||||
GPIO_PORT1_OFFSET = 0x1814,
|
||||
|
||||
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
|
||||
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
|
||||
HW_GP0_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT0_OFFSET,
|
||||
HW_GP1_ADDR = (IO_BASE_ADDR_HI16 << 16) | GPIO_PORT1_OFFSET,
|
||||
};
|
||||
|
||||
#define HW_GP0 C_(U4 V_*, HW_GP0_ADDR)
|
||||
@@ -73,66 +73,64 @@ enum {
|
||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||
* ============================================================================
|
||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
||||
* The layer-1 bitfield-layout constants live in the same enum block so the encoder can reference them by name.
|
||||
* NO macro body past this point uses a raw shift or raw mask.
|
||||
* Every shift/width/mask is named here, named once.
|
||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp0_cmd_Nop = 0x00,
|
||||
gp0_cmd_Nop = 0x00,
|
||||
|
||||
/* Cache management */
|
||||
gp0_cmd_ClearCache = 0x01,
|
||||
gp0_cmd_FillVram = 0x02,
|
||||
gp0_cmd_CopyVram = 0x80,
|
||||
gp0_cmd_CopyVramChained = 0x81,
|
||||
gp0_cmd_ReadVram = 0xC0,
|
||||
/* Cache management */
|
||||
gp0_cmd_ClearCache = 0x01,
|
||||
gp0_cmd_FillVram = 0x02,
|
||||
gp0_cmd_CopyVram = 0x80,
|
||||
gp0_cmd_CopyVramChained = 0x81,
|
||||
gp0_cmd_ReadVram = 0xC0,
|
||||
|
||||
/* Polygons */
|
||||
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
||||
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
||||
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
||||
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
||||
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
||||
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
||||
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
||||
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
||||
/* Polygons */
|
||||
gp0_cmd_poly_f3 = 0x20, /* Flat Triangle */
|
||||
gp0_cmd_poly_ft3 = 0x24, /* Flat Textured Triangle */
|
||||
gp0_cmd_poly_g3 = 0x30, /* Gouraud Triangle */
|
||||
gp0_cmd_poly_gt3 = 0x34, /* Gouraud Textured Tri */
|
||||
gp0_cmd_poly_f4 = 0x28, /* Flat Quad */
|
||||
gp0_cmd_poly_ft4 = 0x2C, /* Flat Textured Quad */
|
||||
gp0_cmd_poly_g4 = 0x38, /* Gouraud Quad */
|
||||
gp0_cmd_poly_gt4 = 0x3C, /* Gouraud Textured Quad */
|
||||
|
||||
/* Lines */
|
||||
gp0_cmd_line_f2 = 0x40,
|
||||
gp0_cmd_line_g2 = 0x50,
|
||||
/* Lines */
|
||||
gp0_cmd_line_f2 = 0x40,
|
||||
gp0_cmd_line_g2 = 0x50,
|
||||
|
||||
/* Sprites + Tiles + Rects */
|
||||
gp0_cmd_sprt_1 = 0x64,
|
||||
gp0_cmd_sprt_8 = 0x74,
|
||||
gp0_cmd_sprt_16 = 0x7C,
|
||||
gp0_cmd_tile_1 = 0x60,
|
||||
gp0_cmd_tile_8 = 0x68,
|
||||
gp0_cmd_tile_16 = 0x70,
|
||||
/* Sprites + Tiles + Rects */
|
||||
gp0_cmd_sprt_1 = 0x64,
|
||||
gp0_cmd_sprt_8 = 0x74,
|
||||
gp0_cmd_sprt_16 = 0x7C,
|
||||
gp0_cmd_tile_1 = 0x60,
|
||||
gp0_cmd_tile_8 = 0x68,
|
||||
gp0_cmd_tile_16 = 0x70,
|
||||
|
||||
/* State setters (not drawing primitives; set render context). */
|
||||
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
|
||||
gp0_cmd_SetTextureWindow = 0xE2,
|
||||
gp0_cmd_SetDrawArea_TopLeft = 0xE3,
|
||||
gp0_cmd_SetDrawArea_BotRight = 0xE4,
|
||||
gp0_cmd_SetDrawOffset = 0xE5,
|
||||
gp0_cmd_SetMaskBit = 0xE6,
|
||||
/* State setters (not drawing primitives; set render context). */
|
||||
gp0_cmd_DrawModeSetting = 0xE1, /* TPage / draw-mode (semi-trans, dither, etc.) */
|
||||
gp0_cmd_SetTextureWindow = 0xE2,
|
||||
gp0_cmd_SetDrawArea_TopLeft = 0xE3,
|
||||
gp0_cmd_SetDrawArea_BotRight = 0xE4,
|
||||
gp0_cmd_SetDrawOffset = 0xE5,
|
||||
gp0_cmd_SetMaskBit = 0xE6,
|
||||
|
||||
/* bitfield shifts / widths / masks ----
|
||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||
gp0_cmd_shift = 24,
|
||||
gp0_cmd_width = 8,
|
||||
gp0_cmd_mask = 0xFF,
|
||||
/* bitfield shifts / widths / masks ----
|
||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||
gp0_cmd_shift = 24,
|
||||
gp0_cmd_width = 8,
|
||||
gp0_cmd_mask = 0xFF,
|
||||
|
||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||
* bits 31..24 = command byte
|
||||
* bits 23..16 = BLUE
|
||||
* bits 15..08 = GREEN
|
||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
||||
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
||||
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||
* bits 31..24 = command byte
|
||||
* bits 23..16 = BLUE
|
||||
* bits 15..08 = GREEN
|
||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
||||
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
||||
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
||||
};
|
||||
|
||||
/* ============================================================================
|
||||
@@ -171,10 +169,13 @@ enum {
|
||||
#define gp0_word_poly_gt4(r,g,b) enc_color_word(gp0_cmd_poly_gt4, (r),(g),(b))
|
||||
|
||||
/* Cache management — bare-cmd words (no color/range payload). */
|
||||
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
||||
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
||||
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
||||
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
||||
#define gp0_word_clear_cache() enc_gp0_cmd_word(gp0_cmd_ClearCache)
|
||||
#define gp0_word_fill_vram() enc_gp0_cmd_word(gp0_cmd_FillVram)
|
||||
#define gp0_word_copy_vram() enc_gp0_cmd_word(gp0_cmd_CopyVram)
|
||||
#define gp0_word_read_vram() enc_gp0_cmd_word(gp0_cmd_ReadVram)
|
||||
|
||||
/* NOP — bare-cmd word (no effect; used as DR_ENV padding). */
|
||||
#define gp0_word_nop() enc_gp0_cmd_word(gp0_cmd_Nop)
|
||||
|
||||
/* ============================================================================
|
||||
* GP1 command byte constants + Layer 1 (display-mode + range + draw-area bitfield shifts)
|
||||
@@ -184,58 +185,57 @@ enum {
|
||||
* (cmd byte in the upper 8 bits via `enc_gp0_cmd(cmd)`).
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp1_cmd_Reset = 0x00,
|
||||
gp1_cmd_ResetCmdBuffer = 0x01,
|
||||
gp1_cmd_AcknowledgeIRQ = 0x02,
|
||||
gp1_cmd_DisplayEnable = 0x03,
|
||||
gp1_cmd_DMADirection = 0x04,
|
||||
gp1_cmd_StartDisplayArea = 0x05,
|
||||
gp1_cmd_HorizontalDisplayRange = 0x06,
|
||||
gp1_cmd_VerticalDisplayRange = 0x07,
|
||||
gp1_cmd_DisplayMode = 0x08,
|
||||
/* Note: GP1 only has commands 0x00..0x08.
|
||||
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
|
||||
* live in the GP0 enum as * 0xE1..0xE6.
|
||||
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
|
||||
gp1_cmd_Reset = 0x00,
|
||||
gp1_cmd_ResetCmdBuffer = 0x01,
|
||||
gp1_cmd_AcknowledgeIRQ = 0x02,
|
||||
gp1_cmd_DisplayEnable = 0x03,
|
||||
gp1_cmd_DMADirection = 0x04,
|
||||
gp1_cmd_StartDisplayArea = 0x05,
|
||||
gp1_cmd_HorizontalDisplayRange = 0x06,
|
||||
gp1_cmd_VerticalDisplayRange = 0x07,
|
||||
gp1_cmd_DisplayMode = 0x08,
|
||||
/* Note: GP1 only has commands 0x00..0x08.
|
||||
* The state-setter commands (SetTextureWindow, * SetDrawArea*, SetDrawOffset, SetMaskBit)
|
||||
* live in the GP0 enum as * 0xE1..0xE6.
|
||||
* DrawArea word builders are below as GP0s * macros (since they emit GP0 commands). */
|
||||
|
||||
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
||||
* Bit positions match the encoder shifts below; values are the
|
||||
* *payload* bits only (the cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
||||
gp1_disp_HRes_256 = 0x0,
|
||||
gp1_disp_HRes_320 = 0x1,
|
||||
gp1_disp_HRes_512 = 0x2,
|
||||
gp1_disp_HRes_640 = 0x3,
|
||||
gp1_disp_VRes_240 = 0x0,
|
||||
gp1_disp_VRes_480 = 0x1,
|
||||
gp1_disp_Color15 = 0x0,
|
||||
gp1_disp_Color24 = 0x1,
|
||||
gp1_disp_VInterlace = 0x1,
|
||||
/* ---- Display-mode payload flags (per PSX-SPX §"GP1 Display Mode").
|
||||
* Bit positions match the encoder shifts below; values are the *payload* bits only (cmd byte is OR'd in by enc_gp1_disp_mode_word). */
|
||||
gp1_disp_HRes_256 = 0x0,
|
||||
gp1_disp_HRes_320 = 0x1,
|
||||
gp1_disp_HRes_512 = 0x2,
|
||||
gp1_disp_HRes_640 = 0x3,
|
||||
gp1_disp_VRes_240 = 0x0,
|
||||
gp1_disp_VRes_480 = 0x1,
|
||||
gp1_disp_Color15 = 0x0,
|
||||
gp1_disp_Color24 = 0x1,
|
||||
gp1_disp_VInterlace = 0x1,
|
||||
|
||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
||||
|
||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
||||
|
||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
||||
|
||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
||||
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
||||
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
||||
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
||||
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask << gp1_disp_interlace_shift)
|
||||
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
||||
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
||||
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
||||
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
|
||||
|
||||
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
|
||||
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
|
||||
@@ -255,6 +255,11 @@ enum {
|
||||
#define enc_gp0_draw_area_br_word(x, y) (enc_gp0_cmd(gp0_cmd_SetDrawArea_BotRight) | enc_gp1_draw_x(x) | enc_gp1_draw_y(y))
|
||||
|
||||
/* ---- Layer 3: GP1 semantic word builders ---- */
|
||||
#define gp1_word_Reset() enc_gp0_cmd_word(gp1_cmd_Reset)
|
||||
#define gp1_word_ResetCmdBuffer() enc_gp0_cmd_word(gp1_cmd_ResetCmdBuffer)
|
||||
#define gp1_word_AcknowledgeIRQ() enc_gp0_cmd_word(gp1_cmd_AcknowledgeIRQ)
|
||||
#define gp1_word_StartDisplayArea() enc_gp0_cmd_word(gp1_cmd_StartDisplayArea)
|
||||
|
||||
#define gp1_word_display_enable(on) (enc_gp0_cmd(gp1_cmd_DisplayEnable) | ((on) & 1))
|
||||
#define gp1_word_display_disable() gp1_word_display_enable(0)
|
||||
#define gp1_word_display_mode_320x240_15bit_ntsc enc_gp1_disp_mode_word(gp1_disp_HRes_320, gp1_disp_VRes_240, gp1_disp_Color15, 0)
|
||||
@@ -279,31 +284,36 @@ enum {
|
||||
#define gp1_word_display_enabled enc_gp0_cmd_word(gp1_cmd_DisplayEnable)
|
||||
#define gp1_word_display_disabled (enc_gp0_cmd_word(gp1_cmd_DisplayEnable) | 1)
|
||||
|
||||
#define gp1_word_DisplayOn() gp1_word_display_enable(0)
|
||||
#define gp1_word_DisplayOff() gp1_word_display_enable(1)
|
||||
|
||||
/* ---- DMA direction (2-bit payload on DMADirection cmd 0x04) ---- */
|
||||
enum {
|
||||
gp1_dma_dir_Off = 0,
|
||||
gp1_dma_dir_FIFO = 1,
|
||||
gp1_dma_dir_CPU_to_GPU = 2,
|
||||
gp1_dma_dir_GPUREAD_to_CPU = 3,
|
||||
gp1_dma_dir_Off = 0,
|
||||
gp1_dma_dir_FIFO = 1,
|
||||
gp1_dma_dir_CPU_to_GPU = 2,
|
||||
gp1_dma_dir_GPUREAD_to_CPU = 3,
|
||||
};
|
||||
#define gp1_word_dma_direction(dir) (enc_gp0_cmd(gp1_cmd_DMADirection) | ((dir) & 0x3))
|
||||
#define gp1_word_dma_to_gpu() gp1_word_dma_direction(gp1_dma_dir_CPU_to_GPU)
|
||||
#define gp1_word_dma_read_cpu() gp1_word_dma_direction(gp1_dma_dir_GPUREAD_to_CPU)
|
||||
|
||||
/* ---- Standard display ranges (NTSC + PAL pre-baked) ---- */
|
||||
/* Horizontal range values are in video clock units (8 units/pixel); vertical range values are scanline numbers. */
|
||||
enum {
|
||||
/* NTSC horizontal range: X1=608, X2=3168 */
|
||||
gp1_hrange_NTSC_x1 = 0x260,
|
||||
gp1_hrange_NTSC_x2 = 0xC60,
|
||||
/* PAL horizontal range (same as NTSC for most CRTs) */
|
||||
gp1_hrange_PAL_x1 = 0x260,
|
||||
gp1_hrange_PAL_x2 = 0xC60,
|
||||
/* NTSC horizontal range: X1=608, X2=3168 */
|
||||
gp1_hrange_NTSC_x1 = 0x260,
|
||||
gp1_hrange_NTSC_x2 = 0xC60,
|
||||
/* PAL horizontal range (same as NTSC for most CRTs) */
|
||||
gp1_hrange_PAL_x1 = 0x260,
|
||||
gp1_hrange_PAL_x2 = 0xC60,
|
||||
|
||||
/* NTSC vertical range: Y1=24, Y2=264 */
|
||||
gp1_vrange_NTSC_y1 = 24,
|
||||
gp1_vrange_NTSC_y2 = 264,
|
||||
/* PAL vertical range: Y1=24, Y2=504 */
|
||||
gp1_vrange_PAL_y1 = 24,
|
||||
gp1_vrange_PAL_y2 = 504,
|
||||
/* NTSC vertical range: Y1=24, Y2=264 */
|
||||
gp1_vrange_NTSC_y1 = 24,
|
||||
gp1_vrange_NTSC_y2 = 264,
|
||||
/* PAL vertical range: Y1=24, Y2=504 */
|
||||
gp1_vrange_PAL_y1 = 24,
|
||||
gp1_vrange_PAL_y2 = 504,
|
||||
};
|
||||
|
||||
#define gp1_word_horizontal_range_ntsc enc_gp1_hrange_word(gp1_hrange_NTSC_x1, gp1_hrange_NTSC_x2)
|
||||
@@ -314,14 +324,49 @@ enum {
|
||||
/* ---- Draw-mode setting (TPage / draw-area allowance) ---- */
|
||||
/* The "drawing enabled" word is the standard post-init state. */
|
||||
enum {
|
||||
gp0_DrawMode_DrawToDispBit = 10,
|
||||
/* Per psx-spx, the standard 0xE1 layout has dfe at bit 10. But libpsyx's PutDrawEnv
|
||||
* uses bit 19 (in the "unused" 14-23 range) for dfe in the DR_ENV code[0] — and the
|
||||
* PSX hardware honors bit 19 in the DR_ENV context (not bit 10). So we need a
|
||||
* separate bit definition for the DR_ENV-specific DrawMode. */
|
||||
gp0_DrawMode_DrawToDispBit = 10, // standard psx-spx bit 10 (dfe)
|
||||
gp0_DrawMode_DR_ENV_DrawToDispBit = 19, // libpsyx DR_ENV code[0] (dfe in DR_ENV context)
|
||||
gp0_DrawMode_DR_ENV_isbgBit = 19, // libpsyx uses bit 19 for isbg too
|
||||
};
|
||||
#define gp0_word_draw_mode_drawing_allowed (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit))
|
||||
|
||||
/* ---- DrawArea pre-baked at origin (0,0) and full screen (320x240) ---- */
|
||||
/* DR_ENV-specific DrawMode variants (libpsyx SetDrawEnv layout).
|
||||
* The DR_ENV is a 16-word packet emitted at boot by gp_screen_init's ac_put_draw_env_demo
|
||||
* atom component. Within the DR_ENV, the 0xE1 command is reused in three different bit
|
||||
* configurations:
|
||||
* code[0] = `gp0_word_draw_mode_drawing_allowed` (dfe=1; standard post-init state)
|
||||
* code[6] = `gp0_word_dr_env_bg_color_cmd(isbg, r, g, b)` (initial-bg-color path)
|
||||
* code[7] = `gp0_word_dr_env_draw_mode(isbg)` (isbg-flag path)
|
||||
* Bits 0-23 of the 0xE1 word are the payload; bits 24-31 are the cmd byte (0xE1). */
|
||||
#define gp0_word_dr_env_bg_color_cmd(isbg, r, g, b) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||
#define gp0_word_dr_env_draw_mode(isbg) (enc_gp0_cmd(gp0_cmd_DrawModeSetting) | (1 << gp0_DrawMode_DrawToDispBit) | ((isbg) ? gp0_dr_env_isbg_bit : 0))
|
||||
|
||||
/* State-setter bare-cmd words (no immediate payload; the GPU uses the current state machine already programmed). */
|
||||
#define gp0_word_set_texture_window() enc_gp0_cmd_word(gp0_cmd_SetTextureWindow)
|
||||
#define gp0_word_set_draw_offset() enc_gp0_cmd_word(gp0_cmd_SetDrawOffset)
|
||||
#define gp0_word_set_mask_bit() enc_gp0_cmd_word(gp0_cmd_SetMaskBit)
|
||||
|
||||
/* DR_ENV code[5] Mask (0xE6 cmd + isbg bit). The isbg bit is set so the GPU knows the auto-clear path is active (paired with code[6] + code[7]). */
|
||||
#define gp0_word_dr_env_mask() (gp0_word_set_mask_bit() | gp0_dr_env_isbg_bit)
|
||||
|
||||
/* DR_ENV pre-baked constants (libpsyx PutDrawEnv layout).
|
||||
* DR_ENV is a 16-word packet: tag = (length << 24) | addr, where length = 15 (15 code words follow) and addr = 0 (chain to nothing). */
|
||||
enum {
|
||||
PolyTag_len_bits = 8,
|
||||
PolyTag_addr_bits = 24,
|
||||
|
||||
gp0_dr_env_tag = (15 << 24) | 0x00FFFFFF,
|
||||
gp0_dr_env_isbg_bit = (1 << gp0_DrawMode_DR_ENV_isbgBit),
|
||||
};
|
||||
|
||||
/* ---- DrawArea at origin (0,0) and full screen (320x240) ---- */
|
||||
#define gp0_word_draw_area_top_left_origin enc_gp0_draw_area_tl_word(0, 0)
|
||||
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(320, 240)
|
||||
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(640, 480)
|
||||
#define gp0_word_draw_area_bottom_right_320x240 enc_gp0_draw_area_br_word(319, 239)
|
||||
#define gp0_word_draw_area_bottom_right_640x480 enc_gp0_draw_area_br_word(639, 479)
|
||||
|
||||
#pragma endregion GPU Ports & Commands
|
||||
|
||||
@@ -332,9 +377,9 @@ enum {
|
||||
* Read from HW_GP1; the lower bits are DMA-block-size (variable-width).
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
gp1_Status_BitReady = 31,
|
||||
gp1_Status_BitSendingDMA = 25,
|
||||
gp1_Status_DMABlockSizeShift = 0,
|
||||
gp1_Status_BitReady = 31,
|
||||
gp1_Status_BitSendingDMA = 25,
|
||||
gp1_Status_DMABlockSizeShift = 0,
|
||||
};
|
||||
|
||||
#define gp1_status_is_ready() ((HW_GP1[0] >> gp1_Status_BitReady) & 1)
|
||||
@@ -360,10 +405,10 @@ typedef Struct_(RGB8) { B1 r; B1 g; B1 b; };
|
||||
#define rgb8(r,g,b) ((RGB8){r,g,b})
|
||||
|
||||
/* ---------- PolyTag (the OT-link header; 1 word) ---------- */
|
||||
enum {
|
||||
PolyTag_len_bits = 8,
|
||||
PolyTag_addr_bits = 24,
|
||||
};
|
||||
// enum {
|
||||
// PolyTag_len_bits = 8,
|
||||
// PolyTag_addr_bits = 24,
|
||||
// };
|
||||
typedef Struct_(PolyTag) {
|
||||
union {
|
||||
U4 code;
|
||||
@@ -387,95 +432,95 @@ typedef Struct_(PolyTag) {
|
||||
|
||||
/* ---------- Poly_F3 (Flat Triangle; 5 words) ---------- */
|
||||
typedef Struct_(Poly_F3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||
A3_V2_S2 points;
|
||||
};
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; };
|
||||
A3_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_F4 (Flat Quad; 6 words) ---------- */
|
||||
typedef Struct_(Poly_F4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||
A4_V2_S2 points;
|
||||
};
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
union {
|
||||
struct { V2_S2 p0; V2_S2 p1; V2_S2 p2; V2_S2 p3; };
|
||||
A4_V2_S2 points;
|
||||
};
|
||||
};
|
||||
|
||||
/* ---------- Poly_G3 (Gouraud Triangle; 7 words) ---------- */
|
||||
typedef Struct_(Poly_G3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_G4 (Gouraud Quad; 9 words) ---------- */
|
||||
typedef Struct_(Poly_G4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
};
|
||||
|
||||
/* ---------- Poly_FT3 (Flat Textured Triangle; placeholder layout) ---------- */
|
||||
/* TODO(Ed): verify the textured-variant layout against PSX-SPX when needed. */
|
||||
typedef Struct_(Poly_FT3) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_FT4 (Flat Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_FT4) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
V2_S2 p3; U1 u3; U1 v3;
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 p0; U1 u0; U1 v0;
|
||||
V2_S2 p1; U1 u1; U1 v1;
|
||||
V2_S2 p2; U1 u2; U1 v2;
|
||||
V2_S2 p3; U1 u3; U1 v3;
|
||||
};
|
||||
|
||||
/* ---------- Poly_GT3 (Gouraud Textured Triangle) ---------- */
|
||||
typedef Struct_(Poly_GT3) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
};
|
||||
|
||||
/* ---------- Poly_GT4 (Gouraud Textured Quad) ---------- */
|
||||
typedef Struct_(Poly_GT4) {
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
V2_S2 tp3; U1 u3; U1 v3;
|
||||
U4 tag; RGB8 c0; B1 code;
|
||||
V2_S2 p0; RGB8 c1; B1 pad1;
|
||||
V2_S2 p1; RGB8 c2; B1 pad2;
|
||||
V2_S2 p2; RGB8 c3; B1 pad3;
|
||||
V2_S2 p3;
|
||||
U4 tpage;
|
||||
U4 clut;
|
||||
V2_S2 tp0; U1 u0; U1 v0;
|
||||
V2_S2 tp1; U1 u1; U1 v1;
|
||||
V2_S2 tp2; U1 u2; U1 v2;
|
||||
V2_S2 tp3; U1 u3; U1 v3;
|
||||
};
|
||||
|
||||
/* ---------- Primitive setters (C-level) ----------
|
||||
@@ -510,26 +555,29 @@ typedef Struct_(Poly_GT4) {
|
||||
* bits 12..31 = reserved (zero)
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
||||
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
||||
|
||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
||||
gp0_tpage_color_4bpp = 0x0,
|
||||
gp0_tpage_color_8bpp = 0x1,
|
||||
gp0_tpage_color_16bpp = 0x2,
|
||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
||||
gp0_tpage_color_4bpp = 0x0,
|
||||
gp0_tpage_color_8bpp = 0x1,
|
||||
gp0_tpage_color_16bpp = 0x2,
|
||||
|
||||
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
||||
gp0_tpage_semi_trans_none = 0x0,
|
||||
gp0_tpage_semi_trans_alpha = 0x1,
|
||||
gp0_tpage_semi_trans_add = 0x2,
|
||||
gp0_tpage_semi_trans_sub = 0x3,
|
||||
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
||||
gp0_tpage_default = 10,
|
||||
|
||||
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
||||
gp0_tpage_semi_trans_none = 0x0,
|
||||
gp0_tpage_semi_trans_alpha = 0x1,
|
||||
gp0_tpage_semi_trans_add = 0x2,
|
||||
gp0_tpage_semi_trans_sub = 0x3,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
||||
@@ -543,19 +591,19 @@ enum {
|
||||
|
||||
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
||||
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||
(enc_gp0_tpage_x(x) \
|
||||
| enc_gp0_tpage_y(y) \
|
||||
| enc_gp0_tpage_semi_trans(semi_trans) \
|
||||
| enc_gp0_tpage_color_depth(color_depth) \
|
||||
| enc_gp0_tpage_dither(dither) \
|
||||
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
||||
| enc_gp0_tpage_tex_disable(tex_disable))
|
||||
(enc_gp0_tpage_x(x) \
|
||||
| enc_gp0_tpage_y(y) \
|
||||
| enc_gp0_tpage_semi_trans(semi_trans) \
|
||||
| enc_gp0_tpage_color_depth(color_depth) \
|
||||
| enc_gp0_tpage_dither(dither) \
|
||||
| enc_gp0_tpage_draw_to_disp(draw_to_disp) \
|
||||
| enc_gp0_tpage_tex_disable(tex_disable))
|
||||
|
||||
typedef Struct_(TexturePage) { U4 raw; };
|
||||
|
||||
/* ---- Layer 3: TPage semantic word builder ---- */
|
||||
#define gp0_word_tpage(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
||||
enc_gp0_tpage_word((x), (y), (semi_trans), (color_depth), (dither), (draw_to_disp), (tex_disable))
|
||||
#pragma endregion TPage
|
||||
|
||||
#pragma region CLUT
|
||||
@@ -569,12 +617,12 @@ typedef Struct_(TexturePage) { U4 raw; };
|
||||
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||
gp0_clut_cmd_Load4bpp = 0x20,
|
||||
gp0_clut_cmd_Load8bpp = 0x25,
|
||||
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||
gp0_clut_cmd_Load4bpp = 0x20,
|
||||
gp0_clut_cmd_Load8bpp = 0x25,
|
||||
};
|
||||
|
||||
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
||||
@@ -615,26 +663,26 @@ enum {
|
||||
* Stoppped for now at the struct + enum level.
|
||||
* ============================================================================ */
|
||||
enum {
|
||||
tim_file_id_magic = 0x10,
|
||||
tim_type_4bpp = 0x00,
|
||||
tim_type_8bpp = 0x01,
|
||||
tim_type_16bpp = 0x02,
|
||||
tim_type_32bpp = 0x03,
|
||||
tim_type_mixed = 0x04,
|
||||
tim_flag_has_clut = 0x08,
|
||||
tim_file_id_magic = 0x10,
|
||||
tim_type_4bpp = 0x00,
|
||||
tim_type_8bpp = 0x01,
|
||||
tim_type_16bpp = 0x02,
|
||||
tim_type_32bpp = 0x03,
|
||||
tim_type_mixed = 0x04,
|
||||
tim_flag_has_clut = 0x08,
|
||||
};
|
||||
|
||||
typedef Struct_(TIM_Header) {
|
||||
U4 file_id; /* always 0x10 = "TIM" magic */
|
||||
U4 version; /* ignored; always 0 */
|
||||
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
||||
U4 file_id; /* always 0x10 = "TIM" magic */
|
||||
U4 version; /* ignored; always 0 */
|
||||
U4 flags; /* bits 0..2 = type, bit 3 = has_clut */
|
||||
};
|
||||
typedef Struct_(TIM_SectionHeader) {
|
||||
U4 section_length; /* bytes in this section including this header */
|
||||
U2 org_x; /* origin in VRAM */
|
||||
U2 org_y;
|
||||
U2 width; /* width in pixels */
|
||||
U2 height; /* height in pixels */
|
||||
U4 section_length; /* bytes in this section including this header */
|
||||
U2 org_x; /* origin in VRAM */
|
||||
U2 org_y;
|
||||
U2 width; /* width in pixels */
|
||||
U2 height; /* height in pixels */
|
||||
};
|
||||
#pragma endregion TIM File Format
|
||||
|
||||
|
||||
+82
-126
@@ -17,9 +17,8 @@
|
||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||
* load_upper_i (load-upper + immediate, unique verb)
|
||||
*
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are
|
||||
* NOT in this header. They live in the opt-in `gte_vendor_sym.h` for
|
||||
* users who prefer the textbook MIPS assembly mnemonics.
|
||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are NOT in this header.
|
||||
* They are in the opt-in `gte_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
|
||||
* ============================================================================ */
|
||||
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
@@ -34,20 +33,16 @@
|
||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||
* ============================================================================
|
||||
*
|
||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word`
|
||||
* constants from C. No GCC inline-assembly string syntax in the code body.
|
||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C.
|
||||
* No GCC inline-assembly string syntax in the code body.
|
||||
*
|
||||
* STYLE NOTES
|
||||
* -----------
|
||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one
|
||||
* self-masks its argument before shifting. Mirrors the `enc_op / enc_rs
|
||||
* / enc_rt / ...` family in mips.h.
|
||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of
|
||||
* the per-field encoders, plus the COP2/CO base.
|
||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined
|
||||
* for the common cases so call sites read like assembly source.
|
||||
* - All register/field values are enums (not `#define`s) so they show up
|
||||
* in debugger symbol tables and IDE autocomplete.
|
||||
* - Per-field encoders are named `enc_gte_<field>(value)` and each one self-masks its argument before shifting.
|
||||
* Mirrors the `enc_op / enc_rs / enc_rt / ...` family in mips.h.
|
||||
* - The composite `enc_gte_cmdw(sf, mx, v, cv, lm, cmd)` is a flat OR of the per-field encoders, plus the COP2/CO base.
|
||||
* - Pre-baked shortcuts (`gte_cmd_rtpt`, `gte_cmd_rtps`, …) are defined for the common cases so call sites read like assembly source.
|
||||
* - All register/field values are enums (not `#define`s) so they show up in debugger symbol tables and IDE autocomplete.
|
||||
*
|
||||
* SEE ALSO
|
||||
* --------
|
||||
@@ -58,8 +53,7 @@
|
||||
|
||||
/* --- GTE Data Registers (Coprocessor 2) ---
|
||||
* Preprocessor-visible integer ids for the COP2 data register file.
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. */
|
||||
#define C2_VXY0_Code 0
|
||||
#define C2_VZ0_Code 1
|
||||
@@ -192,10 +186,8 @@ enum {
|
||||
|
||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the
|
||||
* preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23
|
||||
* are reserved/unused on real hardware, so there's a gap. */
|
||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||
* Same pattern as the GPR `_Code` set in mips.h. Note: indices 21-23 are reserved/unused on real hardware, so there's a gap. */
|
||||
#define gte_cr_RT11_Code 0
|
||||
#define gte_cr_RT12_Code 1 /* packed with RT13 in bits 16..31 */
|
||||
#define gte_cr_RT13_Code 2 /* packed with RT22 in bits 16..31 */
|
||||
@@ -223,8 +215,9 @@ enum {
|
||||
#define gte_cr_RFC_Code 27
|
||||
#define gte_cr_GFC_Code 28
|
||||
#define gte_cr_BFC_Code 29
|
||||
#define gte_cr_OFX_Code 30
|
||||
#define gte_cr_OFY_Code 31
|
||||
#define gte_cr_OFX_Code 24
|
||||
#define gte_cr_OFY_Code 25
|
||||
#define gte_cr_H_Code 26
|
||||
|
||||
enum {
|
||||
gte_cr_RT11 = gte_cr_RT11_Code, gte_cr_RT12 = gte_cr_RT12_Code, gte_cr_RT13 = gte_cr_RT13_Code,
|
||||
@@ -246,21 +239,16 @@ enum { _C2_OPS_ = 0
|
||||
|
||||
/* COP2 transfer sub-opcodes (5-bit field in the `rs` slot of enc_gte_tx).
|
||||
*
|
||||
* Spans the 2x2 {From, To} × {Data, Control} register classes that the
|
||||
* GTE exposes:
|
||||
*
|
||||
* Spans the 2x2 {From, To} × {Data, Control} register classes that the GTE exposes:
|
||||
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||
*
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit
|
||||
* numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* (which target the data register file on any coprocessor). They are
|
||||
* re-aliased here so the four-way table reads like the spec mnemonics
|
||||
* (MFC2 / CFC2 / MTC2 / CTC2) and so the encoding lives next to its
|
||||
* only consumer (this header).
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* (which target the data register file on any coprocessor).
|
||||
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
||||
* and so the encoding lives next to its only consumer (this header).
|
||||
*
|
||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2)
|
||||
* live in gte_vendor_sym.h. */
|
||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
||||
enum { _C2_TX_SUBS_ = 0
|
||||
, sub_mfc2 = 0x00 /* MFC2: Move From Coprocessor 2 data reg */
|
||||
, sub_cfc2 = 0x02 /* CFC2: Copy From Coprocessor 2 ctrl reg */
|
||||
@@ -270,11 +258,11 @@ enum { _C2_TX_SUBS_ = 0
|
||||
|
||||
/* COP2 (GTE) Transfer Format: mfc2 / cfc2 / mtc2 / ctc2 rt, rd
|
||||
* Layout: [op_cop2:6][sub:5][rt:5][rd:5][0:11]
|
||||
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
|
||||
* - rt: GPR source/dest
|
||||
* - rd: COP2 register index (0..31):
|
||||
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
|
||||
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
|
||||
* - sub: one of sub_mfc2 / sub_cfc2 / sub_mtc2 / sub_ctc2
|
||||
* - rt: GPR source/dest
|
||||
* - rd: COP2 register index (0..31):
|
||||
* data class → C2_VXY0_Code..C2_LZCR_Code (gte_in_v0_xy..gte_math_accum2 aliases)
|
||||
* ctrl class → gte_cr_RT11_Code..gte_cr_OFY_Code */
|
||||
#define enc_gte_tx(sub, rt, rd) (enc_op(op_cop2) | enc_rs(sub) | enc_rt(rt) | enc_rd(rd))
|
||||
|
||||
|
||||
@@ -314,8 +302,8 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
||||
* gte_lw rt, base, off → lwc2 rt, off(base)
|
||||
* gte_sw rt, base, off → swc2 rt, off(base)
|
||||
* For the typical user-facing vector-level load (xy + z as two
|
||||
* instructions), use the higher-level `gte_load_vN` macros below. */
|
||||
* For the typical user-facing vector-level load (xy + z as two instructions),
|
||||
* use the higher-level `gte_load_vN` macros below. */
|
||||
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
||||
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
||||
|
||||
@@ -323,13 +311,12 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||
* The lower 25 bits are the GTE-specific command payload.
|
||||
*
|
||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs`
|
||||
* pattern in mips.h: each one self-masks and shifts its own field, so a
|
||||
* caller can build up a GTE command piece by piece (handy for state-driven
|
||||
* MVMVA emitters that vary one field at a time).
|
||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
||||
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
||||
*
|
||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command
|
||||
* word in one go. It just ORs the per-field encoders together. */
|
||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go.
|
||||
* It just ORs the per-field encoders together. */
|
||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||
|
||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||
@@ -359,12 +346,13 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* Decomposition (per the `enc_gte_<field>` definitions above):
|
||||
* gte_cmdw_<name> = gte_cmd_base | enc_gte_cmd(<cmd>)
|
||||
|
||||
* The SF/MX/V/CV/LM fields are all zero in the common cases
|
||||
* The SF / MX / V / CV / LM fields are all zero in the common cases
|
||||
* (standard rotation-matrix, no scaling factor, V0 vector, translation vector, no clamp),
|
||||
* so the only varying bits are the `cmd` field.
|
||||
*
|
||||
* Naming follows the file's convention: `gte_cmd_*` is the raw 6-bit `cmd` field id, `gte_cmdw_*`
|
||||
* is the fully-encoded 32-bit instruction word ready to drop into a `.word` directive.
|
||||
* Naming convention:
|
||||
* - `gte_cmd_*` : Raw 6-bit `cmd` field id
|
||||
* - `gte_cmdw_* : 32-bit instruction word ready to drop into a `.word` directive.
|
||||
*
|
||||
* --------------------------------------------------------------------------
|
||||
* PsyQ-compatibility note (RTPS/RTPT):
|
||||
@@ -380,7 +368,7 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* `nclip` ends up wrong, and the triangle is culled.
|
||||
*
|
||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
|
||||
* NCLIP/OP/MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||
* --------------------------------------------------------------------------
|
||||
*/
|
||||
#define gte_cmdw_psyq_compat (1u << 21 | enc_gte_sf(gte_sf_integer))
|
||||
@@ -413,20 +401,16 @@ enum { _C2_TX_SUBS_ = 0
|
||||
|
||||
/**
|
||||
* @brief Loads a single SVECTOR to GTE vector register V0
|
||||
*
|
||||
* @details Loads values from an SVECTOR struct to GTE data registers C2_VXY0
|
||||
* (XY at offset 0) and C2_VZ0 (Z at offset 4) using `lwc2`.
|
||||
*
|
||||
* Uses string-style GCC inline asm with `%0` substitution because the
|
||||
* base register `r0` is a runtime GPR chosen by the compiler.
|
||||
* Uses string-style GCC inline asm with `%0` substitution because the base register `r0` is a runtime GPR chosen by the compiler.
|
||||
* It cannot be encoded into a static `.word` constant.
|
||||
*
|
||||
* Usage:
|
||||
* asm_gte_load_v0(svector_ptr);
|
||||
* Usage: asm_gte_load_v0(svector_ptr);
|
||||
*/
|
||||
|
||||
/* lwc2 encoding helpers parameterized on the base GPR.
|
||||
*
|
||||
* gte_lw_v0_xy(base) → lwc2 $0, 0(base) ; C2_VXY0
|
||||
* gte_lw_v0_z(base) → lwc2 $1, 4(base) ; C2_VZ0
|
||||
* gte_lw_v1_xy(base) → lwc2 $2, 0(base) ; C2_VXY1
|
||||
@@ -435,8 +419,7 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* gte_lw_v2_z(base) → lwc2 $5, 4(base) ; C2_VZ2
|
||||
*
|
||||
* `base` is the GPR number to bake into the .word constant's `rs` field.
|
||||
* These are pure compile-time integers; the C compiler constant-folds
|
||||
* them into .word directives. */
|
||||
* These are pure compile-time integers; the C compiler constant-folds them into .word directives. */
|
||||
|
||||
enum {
|
||||
GTE_Z_Offset = 4
|
||||
@@ -459,8 +442,8 @@ enum {
|
||||
* gte_load_v0(p_in_12, R_T4); // R_T4 = 12, base is $12
|
||||
*
|
||||
* Then `"r"(r_ptr)` inside the asm binds to $12 (the only register `p_in_12` can live in),
|
||||
* which is exactly the register the .word constants expect. A `"$12"` clobber would conflict with the register-variable binding
|
||||
* ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
|
||||
* which is exactly the register the .word constants expect.
|
||||
* A `"$12"` clobber would conflict with the register-variable binding ("asm specifier for variable conflicts with asm clobber list"), so we omit it.
|
||||
* The other ABI-clobbers ($2/$8/$9/$31) stay because the GTE instructions don't touch caller-saved GPRs but the kernel does treat them as volatile.
|
||||
*
|
||||
* WHICH REGISTER TO PICK
|
||||
@@ -499,10 +482,8 @@ enum {
|
||||
|
||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
||||
*
|
||||
* Loads all three GTE input vectors (6 words) from three separate pointers,
|
||||
* one per GTE vector register, each loaded from its own base GPR.
|
||||
* Caller must bind each `pN` to `bN` via a register variable.
|
||||
*
|
||||
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register,
|
||||
* each loaded from its own base GPR. Caller must bind each `pN` to `bN` via a register variable.
|
||||
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
||||
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
||||
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
||||
@@ -521,29 +502,20 @@ enum {
|
||||
|
||||
/**
|
||||
* @brief Rotate, Translate and Perspective Triple (23 cycles)
|
||||
*
|
||||
* @details Performs rotation, translation and perspective calculation of three
|
||||
* vertices at once. The equation performed is the same as gte_rtps() only
|
||||
* repeated three times for each vertex. The result of the first vertex is
|
||||
* stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then
|
||||
* C2_SXY2.
|
||||
* @details Performs rotation, translation and perspective calculation of three vertices at once.
|
||||
* The equation performed is the same as gte_rtps() only repeated three times for each vertex.
|
||||
* The result of the first vertex is stored in GTE data register C2_SXY0, the second vector in C2_SXY1 then C2_SXY2.
|
||||
*
|
||||
* Encoder-style emission (no inline-asm strings in the code body):
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE
|
||||
* takes ~8 cycles per perspective divide, and the nops let any
|
||||
* preceding lwc2/swc2 retire before RTPT starts reading its
|
||||
* inputs from V0/V1/V2.
|
||||
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the
|
||||
* pre-baked encoders above) — `0x0280030` decoded as
|
||||
* `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields
|
||||
* zero (standard rotation, no scaling, V0 vector, translation
|
||||
* vector, no clamp).
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency — the GTE takes ~8 cycles per perspective divide,
|
||||
* and the nops let any preceding lwc2/swc2 retire before RTPT starts reading its inputs from V0/V1/V2.
|
||||
* 2. The RTPT command word itself is `gte_cmdw_rtpt` (see the pre-baked encoders above) —
|
||||
* `0x0280030` decoded as `op_cop2` | CO(1) | cmd=RTPT, with all SF/MX/V/CV/LM fields zero
|
||||
* (standard rotation, no scaling, V0 vector, translation vector, no clamp).
|
||||
*
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
||||
* data/control register — those have to be saved by the caller if
|
||||
* they need to survive across the call (RTPT writes SXY0..2, SZ0..3,
|
||||
* OTZ, MAC0..3, IR0..3, etc.).
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI)
|
||||
* plus the standard "memory" barrier. Does not clobber any COP2 data/control register —
|
||||
* those have to be saved by the caller if they need to survive across the call (RTPT writes SXY0..2, SZ0..3, OTZ, MAC0..3, IR0..3, etc.).
|
||||
*/
|
||||
#define gte_rtpt() \
|
||||
asm volatile( \
|
||||
@@ -559,32 +531,24 @@ enum {
|
||||
|
||||
/**
|
||||
* @brief Normal clipping (8 cycles)
|
||||
*
|
||||
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for
|
||||
* backface culling. If the value of C2_MAC0 is negative, the coordinates are
|
||||
* inverted and thus the triangle is back facing.
|
||||
* @details Computes the sign of three screen coordinates (C2_SXY0-2) used for backface culling.
|
||||
* If the value of C2_MAC0 is negative, the coordinates are inverted and thus the triangle is back facing.
|
||||
*
|
||||
* The following equation is performed when executing this GTE command:
|
||||
*
|
||||
* MAC0 = SX0*SY1 + SX1*SY2 + SX2*SY0 - SX0*SY2 - SX1*SY0 - SX2*SY1
|
||||
*
|
||||
* Encoder-style emission (no inline-asm strings in the code body):
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency - the GTE
|
||||
* pipeline takes a few cycles per op, and the nops let any
|
||||
* preceding lwc2/swc2/RTPT retire before NCLIP starts reading
|
||||
* its inputs from SXY0/SXY1/SXY2.
|
||||
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the
|
||||
* pre-baked encoders above) - `0x01400006` decoded as
|
||||
* `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields
|
||||
* zero. NCLIP is spec-clean in the original PsyQ source
|
||||
* (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat`
|
||||
* quirk), so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
||||
* 1. Two `nop` words fill the COP2 pipeline latency
|
||||
* - the GTE pipeline takes a few cycles per op, and the nops let any preceding
|
||||
* lwc2/swc2/RTPT retire before NCLIP starts reading its inputs from SXY0/SXY1/SXY2.
|
||||
* 2. The NCLIP command word itself is `gte_cmdw_nclip` (see the pre-baked encoders above)
|
||||
* - `0x01400006` decoded as `op_cop2` | CO(1) | cmd=NCLIP, with all SF/MX/V/CV/LM fields zero.
|
||||
* NCLIP is spec-clean in the original PsyQ source (unlike RTPS/RTPT which carry the `gte_cmdw_psyq_compat` quirk),
|
||||
* so `gte_cmdw_nclip` does NOT OR in any reserved bits.
|
||||
*
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel
|
||||
* ABI) plus the standard "memory" barrier. Does not clobber any COP2
|
||||
* data/control register - those have to be saved by the caller if
|
||||
* they need to survive across the call (NCLIP writes MAC0 only; it
|
||||
* is purely a sign-of-double-product computation on SXY0..2).
|
||||
* Clobbers the caller-saved GPRs via `clbr_volatile_gprs` (per the kernel ABI) plus the standard "memory" barrier.
|
||||
* Does not clobber any COP2 data/control register.
|
||||
* Those have to be saved by the caller if they need to survive across the call (NCLIP writes MAC0 only;
|
||||
* it is purely a sign-of-double-product computation on SXY0..2).
|
||||
*/
|
||||
#define gte_nclip() \
|
||||
asm volatile( \
|
||||
@@ -610,13 +574,10 @@ enum {
|
||||
"cop2 0x0158002D;")
|
||||
|
||||
/* asm_gte_matrix_set_rotation(r0)
|
||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||
*
|
||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix
|
||||
* control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||
*
|
||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16),
|
||||
* each holding two packed 16-bit matrix elements. The first 1.5 rows
|
||||
* of a standard PSX SDK MATRIX struct (where each row is laid out as
|
||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
||||
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as
|
||||
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||
*
|
||||
* Generated MIPS (mirrors the source macro):
|
||||
@@ -631,27 +592,22 @@ enum {
|
||||
* ctc2 $13, $3 ; → C2_RT21
|
||||
* ctc2 $14, $4 ; → C2_RT22
|
||||
*
|
||||
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a
|
||||
* register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
||||
* instructions to read from the right base. The `"r"(r0)` constraint
|
||||
* alone doesn't force a specific GPR — it just lets GCC pick one.
|
||||
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field
|
||||
* of each lw, so the lw instructions will only do the right thing
|
||||
* if $12/$13/$14 hold the matrix base at runtime.
|
||||
* Same contract as gte_load_v0: caller MUST bind `r0` to $12 via a register variable (`rgcc(R_T4)`) for the `lw $12, off(...)`
|
||||
* instructions to read from the right base. The `"r"(r0)` constraint alone doesn't force a specific GPR — it just lets GCC pick one.
|
||||
* The .word constants here bake R_T4/R_T5/R_T6 into the `rs` field of each lw, so the lw instructions will
|
||||
* only do the right thing if $12 / $13 / $14 hold the matrix base at runtime.
|
||||
*
|
||||
* M3_S2* m = ...;
|
||||
* register M3_S2* m_in_12 rgcc(R_T4) = m;
|
||||
* asm_gte_matrix_set_rotation(m_in_12);
|
||||
*
|
||||
* We clobber $12/$13/$14 (the ones we use as scratch inside the
|
||||
* inline asm) plus the system clobbers; we don't clobber `r0` because
|
||||
* the `rgcc` binding already says "this variable lives in $12".
|
||||
* We clobber $12/$13/$14 (the ones we use as scratch inside the inline asm)
|
||||
* plus the system clobbers; we don't clobber `r0` because the `rgcc` binding already says "this variable lives in $12".
|
||||
*
|
||||
* WARNING: Incomplete by design. The source macro only writes RT11..RT22
|
||||
* (5 of 9 rotation elements); RT23 and the entire RT3x row are left
|
||||
* untouched. Real libpsn00b SetRotMatrix writes all 9. Use only when the
|
||||
* GTE's remaining rotation entries are already correct, or you will
|
||||
* get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
||||
* WARNING: Incomplete by design. The source macro only writes RT11..RT22 (5 of 9 rotation elements);
|
||||
* RT23 and the entire RT3x row are left untouched.
|
||||
* Real libpsn00b SetRotMatrix writes all 9. Use only when the GTE's remaining rotation entries are already correct,
|
||||
* or you will get stale-RT2x/RT3x artifacts in RTPS/RTPT/MVMVA output.
|
||||
*/
|
||||
#define asm_gte_matrix_set_rotation(r0) \
|
||||
asm volatile( \
|
||||
|
||||
+81
-118
@@ -10,10 +10,10 @@
|
||||
# include "gen/duffle.offsets.h"
|
||||
#endif
|
||||
|
||||
typedef U4 const MipsCode;
|
||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||
typedef Slice_(MipsCode);
|
||||
typedef Slice_MipsCode MipsAtom;
|
||||
|
||||
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
|
||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||
@@ -23,63 +23,78 @@ typedef Slice_MipsCode MipsAtom;
|
||||
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for components with value-args (e.g., ac_format_f3_color).
|
||||
// FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||
// FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||
// expands to:
|
||||
// FI_ MipsAtom ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
||||
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
||||
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
|
||||
|
||||
// Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h) are included manually by the unity build.
|
||||
|
||||
/* Register aliases */
|
||||
enum {
|
||||
R_AtomJmp = R_T9 atom_reg, /* debug-visible; tape yield handshake scratch */
|
||||
R_TapePtr = R_T8 atom_reg, /* The Instruction Stream Pointer */
|
||||
R_InCursor = R_T4,
|
||||
|
||||
R_PrimCursor = R_T7 atom_reg atom_type(U4 *), /* VRAM output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2 *), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2 *), /* Base address of the vertex array */
|
||||
R_OtBase = R_T6 atom_reg atom_type(U4 *), /* Base address of the Ordering Table */
|
||||
|
||||
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
|
||||
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
|
||||
/* Stringification codes for the GCC inline assembler clobber lists. */
|
||||
#define R_TapePtr_Code R_T8_Code
|
||||
#define R_InCursor_Code R_T4_Code
|
||||
#define R_AtomJmp_Code R_T8_Code
|
||||
#define R_TapePtr_Code R_T9_Code
|
||||
|
||||
#define R_PrimCursor_Code R_T7_Code
|
||||
#define R_FaceCursor_Code R_T4_Code
|
||||
#define R_VertBase_Code R_T5_Code
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
// R_InCursor = R_T4,
|
||||
// #define R_InCursor_Code R_T4_Code
|
||||
|
||||
// Reserved Registers (Callee-saved):
|
||||
// - R_T9: Holds the Tape Ptr which we need to increment
|
||||
// If we hit a wall with register allocations we can clobber V0 & V1 (return values), defering as opt-in by user.
|
||||
// - R_RA: Not sure??
|
||||
// Needed by ac_yield but can be used as atom scratch:
|
||||
// - R_T8: Will be used as the atom jump register.
|
||||
|
||||
// All allocatable registers for mips atoms:
|
||||
R_TScratchVolatile = R_AT, // This one is reserved for psuedo instructions, but you can technically use it.
|
||||
R_TScratch0 = R_T0,
|
||||
R_TScratch1 = R_T1,
|
||||
R_TScratch2 = R_T2,
|
||||
R_TScratch3 = R_T3,
|
||||
R_TScratch4 = R_T4,
|
||||
R_TScratch5 = R_T5,
|
||||
R_TScratch6 = R_T6,
|
||||
R_TScratch7 = R_T7,
|
||||
R_TScratch8 = R_T8,
|
||||
R_TScratch10 = R_V0,
|
||||
R_TScratch11 = R_V1,
|
||||
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
|
||||
// R_TScratch12 = R_A0,
|
||||
// R_TScratch13 = R_A1,
|
||||
// R_TScratch14 = R_A3,
|
||||
// TODO(Ed): Review S0-S7, they are technically avaialble, we just have to snapshot them at the ABI boundary.
|
||||
// TODO(Ed): This is technically a waste of cycles for most work? so maybe only do this for expensive atoms on-demand or atom phases.
|
||||
// TODO(Ed): Sort out the other available registers... (Not sure how much is left avail)
|
||||
};
|
||||
|
||||
#pragma region Tape Drive
|
||||
/* ---------------------------------------------------------------------------
|
||||
* TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below)
|
||||
* ---------------------------------------------------------------------------*/
|
||||
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
|
||||
|
||||
/* The 'Exit' Atom */
|
||||
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
||||
|
||||
//TODO(Ed): Do we backup R_S0-7 here? Have it in a heavier tape run as a opt-in? Same with V0-1 and A0-3?
|
||||
/* Generalized Tape Engine Runner */
|
||||
NI_ void tape_run(Slice_MipsCode tape) { register U4* tp rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||
asm_words(
|
||||
add_ui( R_SP, R_SP, -MipsStackAlignment) /* Allocate stack space */
|
||||
, store_word( R_RA, R_SP, 0) /* Safely backup $ra to the stack */
|
||||
, load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||
, add_ui_self(R_TapePtr, S_(MipsCode)) /* Advance tape */
|
||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||
, nop /* Branch delay slot */
|
||||
, load_word( R_RA, R_SP, 0) /* Restore $ra from stack */
|
||||
, add_ui_self(R_SP, MipsStackAlignment) /* Deallocate stack space */
|
||||
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
|
||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||
, nop /* Branch delay slot */
|
||||
)
|
||||
asm_rpins, r_use(tp)
|
||||
asm_rpins, r_use(tape_ptr)
|
||||
asm_clobber:
|
||||
rlit(R_AT)
|
||||
, rlit(R_V0), rlit(R_V1)
|
||||
, rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3)
|
||||
/* Tell GCC the tape engine owns and destroys the workspace registers */
|
||||
, rlit(R_PrimCursor), rlit(R_FaceCursor), rlit(R_VertBase), rlit(R_OtBase)
|
||||
, rlit(R_T9)
|
||||
, clb_mem_drain
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
||||
clb_mem_drain
|
||||
); }
|
||||
|
||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||
@@ -87,13 +102,17 @@ FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start
|
||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
||||
|
||||
#define tb_emit_(tb, atom) tb_emit(tb, atom)
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
||||
|
||||
FI_ Slice_MipsCode tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Slice_MipsCode){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Slice_MipsCode tb_slice(TapeBuilder tb) { return (Slice_MipsCode){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
|
||||
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
|
||||
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
|
||||
|
||||
#pragma endregion Tape Drive
|
||||
|
||||
@@ -110,6 +129,17 @@ atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||
jump_reg( R_AtomJmp), nop,
|
||||
};
|
||||
|
||||
enum {
|
||||
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||
#define R_PrimCursor_Code R_T7_Code
|
||||
#define R_FaceCursor_Code R_T4_Code
|
||||
#define R_VertBase_Code R_T5_Code
|
||||
#define R_OtBase_Code R_T6_Code
|
||||
};
|
||||
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) {
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
@@ -156,7 +186,7 @@ MipsAtomComp_(ac_insert_ot_tag_g4) {
|
||||
|
||||
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
|
||||
* byte offset. Internal helper used by the *_format_*_color macros. */
|
||||
FI_ MipsAtom ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||
FI_ Slice_MipsCode ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
|
||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||
@@ -165,12 +195,12 @@ atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
|
||||
|
||||
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
||||
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
|
||||
FI_ MipsAtom ac_format_f3_color(U1 r, U1 g, U1 b)
|
||||
FI_ Slice_MipsCode ac_format_f3_color(U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_f3) {
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
|
||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
|
||||
@@ -178,7 +208,7 @@ atom_dbg_skip MipsAtomComp_(ac_gte_store_f3_post_rtpt) {
|
||||
|
||||
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
|
||||
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
|
||||
FI_ MipsAtom ac_format_g4_color(
|
||||
FI_ Slice_MipsCode ac_format_g4_color(
|
||||
U1 r0, U1 g0, U1 b0,
|
||||
U1 r1, U1 g1, U1 b1,
|
||||
U1 r2, U1 g2, U1 b2,
|
||||
@@ -193,24 +223,19 @@ MipsAtomComp_Proc_(ac_format_g4_color, {
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||
* G4 triangle portion to p0/p1/p2.
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2
|
||||
* get overwritten with v3 (RTPS writes only to SXY2, but to keep the
|
||||
* three registers aligned with v0/v1/v2 you must store before RTPS).
|
||||
* The macro name declares the pipeline position; check #6 (GTE state-
|
||||
* machine validation) verifies the call site matches the declaration. */
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012_post_rtpt_pre_rtps) {
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012) {
|
||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
|
||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
|
||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
|
||||
};
|
||||
|
||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its
|
||||
* single-vertex result to SXY2; SXY0 still holds v0.screen from the
|
||||
* earlier RTPT — DO NOT read SXY0 here, that's the bug this name
|
||||
* prevents).
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||
*/
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3_post_rtps) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
|
||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
|
||||
|
||||
#pragma endregion Macro Atom Components
|
||||
|
||||
@@ -237,7 +262,7 @@ FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
|
||||
}
|
||||
|
||||
#define mipsatom_from_builder(ab) (MipsAtom){ab.start, ab.used}
|
||||
#define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used}
|
||||
|
||||
#pragma endregion Mips Atom Builder
|
||||
|
||||
@@ -291,66 +316,4 @@ internal MipsAtom_(set_gte_world) atom_info(
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||
|
||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||
internal MipsAtom_(diag_color) {
|
||||
store_word( R_0, R_T7, 0),
|
||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||
store_word( R_AT, R_T7, 4),
|
||||
|
||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||
|
||||
add_ui( R_T1, R_0, 10),
|
||||
shift_lleft_self(R_T1, S_(U4)/2),
|
||||
add_u_self( R_T1, R_T6),
|
||||
|
||||
load_word( R_AT, R_T1, 0),
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
store_word( R_AT, R_T7, 0),
|
||||
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
or_u_self( R_AT, R_V0),
|
||||
store_word( R_AT, R_T1, 0),
|
||||
|
||||
add_ui(R_T7, R_T7, 20),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||
internal MipsAtom_(diag_gte) {
|
||||
/* Load 3 indices */
|
||||
load_half_u(R_T0, R_T4, 0),
|
||||
load_half_u(R_T1, R_T4, 2),
|
||||
load_half_u(R_T2, R_T4, 4),
|
||||
|
||||
/* Load Vertices into GTE */
|
||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
|
||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
|
||||
/* Run Math */
|
||||
nop2, gte_cmdw_rtpt,
|
||||
nop2, gte_cmdw_nclip,
|
||||
nop2,
|
||||
|
||||
/* Advance Face Cursor and Yield */
|
||||
add_ui(R_T4, R_T4, 8),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Mips Atoms
|
||||
|
||||
@@ -11,6 +11,7 @@ enum {
|
||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||
};
|
||||
|
||||
typedef Array_(U1, 2);
|
||||
typedef Array_(U4, 2);
|
||||
typedef Array_(S2, 2);
|
||||
typedef Array_(S2, 3);
|
||||
@@ -22,6 +23,7 @@ typedef S2 A3x3_S2[3][3];
|
||||
typedef Struct_(Extent2_S2) { S2 width; S2 height; };
|
||||
typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
||||
|
||||
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
||||
@@ -37,6 +39,7 @@ typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||
|
||||
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
||||
|
||||
typedef Array_(V2_S2, 2);
|
||||
typedef Array_(V2_S2, 3);
|
||||
typedef Array_(V2_S2, 4);
|
||||
|
||||
|
||||
@@ -67,8 +67,8 @@ typedef Slice_(B1);
|
||||
#define slice_end(slice) ((slice).ptr + (slice).len)
|
||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||
|
||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||
#define slice_ut_arr(a) slice_ut_(u4_(a), S_(a))
|
||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||
|
||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||
|
||||
+6
-9
@@ -336,10 +336,10 @@ enum { _BitOffsets = 0
|
||||
|
||||
/* Logic Opcodes */
|
||||
|
||||
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
||||
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
||||
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
||||
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
||||
#define and_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_and)
|
||||
#define or_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_or)
|
||||
#define xor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_xor)
|
||||
#define nor_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_nor)
|
||||
|
||||
#define or_u_self(rd_rs, rt) enc_r(op_special, (rd_rs), (rt), (rd_rs), 0, fc_or)
|
||||
|
||||
@@ -444,18 +444,15 @@ enum { _BitOffsets = 0
|
||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||
|
||||
/* load_imm_2w — unconditional 2-word `li` form: `lui` + (ori | addi).
|
||||
*
|
||||
* Granular companion to `load_imm`: skips the compile-time range checks and always emits 2 .words. Use this when:
|
||||
* - you know `imm` is > 0xFFFF (otherwise you're wasting a word), OR
|
||||
* - `imm` is not a compile-time constant and you want predictable
|
||||
* 2-word emission without the `__builtin_constant_p` branches.
|
||||
* - `imm` is not a compile-time constant and you want predictable 2-word emission without the `__builtin_constant_p` branches.
|
||||
*
|
||||
* The lo16 strategy is still chosen at expansion time on the lo half:
|
||||
* lo16 in 0x0000..0x7FFF → addi (sign-ext is harmless, the lui already cleared bits 15..0)
|
||||
* lo16 in 0x8000..0xFFFF → ori (zero-extends to preserve the intended bit pattern)
|
||||
*
|
||||
* For situations where you need to bypass even this choice
|
||||
* (e.g. to force a specific encoding for a known discontiguous high/low pair),
|
||||
* For situations where you need to bypass even this choice (e.g. to force a specific encoding for a known discontiguous high/low pair),
|
||||
* see `load_imm_2w_ori_forced` and `load_imm_2w_addi_forced` below.
|
||||
* Statement-level (not expression-level): emits its own `asm volatile(...)`.
|
||||
*/
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
#endif
|
||||
|
||||
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||
* Wire is active-low (0 = pressed).
|
||||
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */
|
||||
enum {
|
||||
Bit_(Pad_Select, 0),
|
||||
Bit_(Pad_L3, 1),
|
||||
Bit_(Pad_R3, 2),
|
||||
Bit_(Pad_Start, 3),
|
||||
Bit_(Pad_Up, 4),
|
||||
Bit_(Pad_Right, 5),
|
||||
Bit_(Pad_Down, 6),
|
||||
Bit_(Pad_Left, 7),
|
||||
Bit_(Pad_L2, 8),
|
||||
Bit_(Pad_R2, 9),
|
||||
Bit_(Pad_L1, 10),
|
||||
Bit_(Pad_R1, 11),
|
||||
Bit_(Pad_Triangle, 12),
|
||||
Bit_(Pad_Circle, 13),
|
||||
Bit_(Pad_Cross, 14),
|
||||
Bit_(Pad_Square, 15),
|
||||
};
|
||||
|
||||
enum {
|
||||
PadId_Offset = 4,
|
||||
|
||||
Pad0 = 0 << PadId_Offset,
|
||||
Pad1 = 1 << PadId_Offset,
|
||||
};
|
||||
|
||||
#define pad0_(btn_id) (btn_id << Pad0)
|
||||
#define pad1_(btn_id) (btn_id << Pad1)
|
||||
|
||||
/* ============================================================
|
||||
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
||||
* ============================================================ */
|
||||
|
||||
enum {
|
||||
PAD_BIOS_RAW_SIZE = 0x22,
|
||||
};
|
||||
typedef Struct_(PadBiosRaw) {
|
||||
U1 bytes[PAD_BIOS_RAW_SIZE];
|
||||
};
|
||||
|
||||
typedef Enum_(U4, PadStatus) {
|
||||
PadStatus_Disconnected,
|
||||
PadStatus_Digital,
|
||||
PadStatus_AnalogStick,
|
||||
PadStatus_AnalogPad,
|
||||
PadStatus_Unsupported,
|
||||
PadStatus_Pending,
|
||||
PadStatus_Invalid,
|
||||
};
|
||||
|
||||
/* PadState — per-port normalized runtime state.
|
||||
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||
* The struct size stays 12 bytes (unchanged from the prior order,
|
||||
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */
|
||||
typedef Struct_(PadState) {
|
||||
PadStatus status; /* offset 0, size 4 (U4) */
|
||||
U2 buttons; /* offset 4, size 2 */
|
||||
U1 id; /* offset 6, size 1 */
|
||||
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
|
||||
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
|
||||
U1 left_y; /* offset 9, size 1 */
|
||||
U1 right_x; /* offset 10, size 1 */
|
||||
U1 right_y; /* offset 11, size 1 */
|
||||
};
|
||||
@@ -22,6 +22,7 @@ WORD_COUNT(call_reg, 1)
|
||||
WORD_COUNT(call_addr, 1)
|
||||
WORD_COUNT(branch_le_zero, 1)
|
||||
WORD_COUNT(branch_equal, 1)
|
||||
WORD_COUNT(branch_ne, 1)
|
||||
WORD_COUNT(add_ui, 1)
|
||||
WORD_COUNT(set_lt_u, 1)
|
||||
WORD_COUNT(set_lt_s, 1)
|
||||
@@ -29,7 +30,9 @@ WORD_COUNT(set_lt_si, 1)
|
||||
WORD_COUNT(set_lt_ui, 1)
|
||||
WORD_COUNT(load_word, 1)
|
||||
WORD_COUNT(load_half_u, 1)
|
||||
WORD_COUNT(load_byte_u, 1)
|
||||
WORD_COUNT(store_word, 1)
|
||||
WORD_COUNT(store_byte, 1)
|
||||
WORD_COUNT(add_ui_self, 1)
|
||||
WORD_COUNT(add_u_self, 1)
|
||||
WORD_COUNT(add_u, 1)
|
||||
@@ -37,6 +40,7 @@ WORD_COUNT(or_i, 1)
|
||||
WORD_COUNT(or_i_self, 1)
|
||||
WORD_COUNT(or_u, 1)
|
||||
WORD_COUNT(or_u_self, 1)
|
||||
WORD_COUNT(nor_u, 1)
|
||||
WORD_COUNT(shift_lleft, 1)
|
||||
WORD_COUNT(shift_lleft_self, 1)
|
||||
WORD_COUNT(shift_lright, 1)
|
||||
|
||||
@@ -1,156 +0,0 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
# include "duffle/atom_dsl.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "gen/gte_hello.offsets.h"
|
||||
# include "hello_gte.h"
|
||||
#endif
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple,
|
||||
nop2, gte_cmdw_nclip,
|
||||
|
||||
nop2, gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
|
||||
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
mac_format_g4_color(
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
mac_gte_store_g4_p012_post_rtpt_pre_rtps(),
|
||||
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
nop2, gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3_post_rtps(),
|
||||
|
||||
nop2, gte_cmdw_avg_sort_z4,
|
||||
nop2, gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(),
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursr)
|
||||
) {
|
||||
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||
/* Format Primitive */
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_gte_store_f3_post_rtpt(),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
/* Insert into Ordering Table Linked List */
|
||||
mac_insert_ot_tag_f3(),
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -5,10 +5,10 @@
|
||||
#pragma region hello_gte_tape
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (87 words) ---
|
||||
// --- atom: cube_g4_face (77 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 48
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 12
|
||||
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
@@ -18,7 +18,7 @@ enum {
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 13
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
@@ -0,0 +1,29 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\hello_gte\hello_gte.tape.c
|
||||
#pragma once
|
||||
|
||||
#pragma region hello_gte.tape
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (77 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||
};
|
||||
|
||||
#pragma endregion hello_gte.tape
|
||||
|
||||
@@ -20,10 +20,10 @@
|
||||
#include "duffle/lottes_tape.h"
|
||||
#include "duffle/word_count.metadata.h"
|
||||
|
||||
# include "gen/gte_hello.offsets.h"
|
||||
# include "gen/hello_gte.offsets.h"
|
||||
#include "hello_gte.h"
|
||||
|
||||
#include "hello_gte_tape.c"
|
||||
#include "hello_gte.tape.c"
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
@@ -122,7 +122,7 @@ global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
// TODO(Ed):
|
||||
FI_ U4* spad_warm(MipsAtom atom) {
|
||||
FI_ U4* spad_warm(Slice_MipsCode atom) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,218 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
# include "duffle/atom_dsl.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "gen/hello_gte.offsets.h"
|
||||
# include "hello_gte.h"
|
||||
#endif
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||
|
||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||
internal MipsAtom_(diag_color) {
|
||||
store_word( R_0, R_T7, 0),
|
||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||
store_word( R_AT, R_T7, 4),
|
||||
|
||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||
|
||||
add_ui( R_T1, R_0, 10),
|
||||
shift_lleft_self(R_T1, S_(U4)/2),
|
||||
add_u_self( R_T1, R_T6),
|
||||
|
||||
load_word( R_AT, R_T1, 0),
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
store_word( R_AT, R_T7, 0),
|
||||
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
or_u_self( R_AT, R_V0),
|
||||
store_word( R_AT, R_T1, 0),
|
||||
|
||||
add_ui(R_T7, R_T7, 20),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||
internal MipsAtom_(diag_gte) {
|
||||
/* Load 3 indices */
|
||||
load_half_u(R_T0, R_T4, 0),
|
||||
load_half_u(R_T1, R_T4, 2),
|
||||
load_half_u(R_T2, R_T4, 4),
|
||||
|
||||
/* Load Vertices into GTE */
|
||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
|
||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
|
||||
/* Run Math */
|
||||
nop2, gte_cmdw_rtpt,
|
||||
nop2, gte_cmdw_nclip,
|
||||
nop2,
|
||||
|
||||
/* Advance Face Cursor and Yield */
|
||||
add_ui(R_T4, R_T4, 8),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
mac_gte_store_g4_p012(),
|
||||
gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3(),
|
||||
|
||||
gte_cmdw_avg_sort_z4,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(),
|
||||
mac_format_g4_color(
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(cull)
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||
/* Format Primitive */
|
||||
mac_gte_store_f3(),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(culling)
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,71 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
#pragma once
|
||||
#endif
|
||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.tape.c
|
||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||
|
||||
#ifndef WORD_COUNT
|
||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||
#endif
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||
load_half( rs_x, r_base, O_(V3_S2,x)) \
|
||||
, load_half( rs_y, r_base, O_(V3_S2,y))
|
||||
WORD_COUNT(mac_load_v2s2, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
|
||||
store_half(rt_x, base, offset + O_(V2_S2,x)) \
|
||||
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
||||
WORD_COUNT(mac_store_v2s2, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
|
||||
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
|
||||
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
||||
WORD_COUNT(mac_store_rects2, 4)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_rgb8(rr, rg, rb, base, offset) \
|
||||
store_byte(rr, base, offset + O_(DrawEnv,initial_bg_color.r)) \
|
||||
, store_byte(rg, base, offset + O_(DrawEnv,initial_bg_color.g)) \
|
||||
, store_byte(rb, base, offset + O_(DrawEnv,initial_bg_color.b))
|
||||
WORD_COUNT(mac_store_rgb8, 3)
|
||||
|
||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||
load_upper_i(reg_transfer, cmd >> 16) \
|
||||
, or_i_self( reg_transfer, cmd & 0xFFFF) \
|
||||
, store_word( reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_gcmd_push, 3)
|
||||
|
||||
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_disp_env, 15)
|
||||
|
||||
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
|
||||
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
|
||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
|
||||
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
|
||||
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_draw_env, 48)
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||
// Source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.tape.c
|
||||
#pragma once
|
||||
|
||||
#pragma region hello_joypad.tape
|
||||
|
||||
|
||||
// --- atom: cube_g4_face (77 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 42
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
enum {
|
||||
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: floor_f3_face (58 words) ---
|
||||
|
||||
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||
|
||||
enum {
|
||||
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||
};
|
||||
|
||||
// --- atom: pad_bios_snapshot (78 words) ---
|
||||
|
||||
#define _atom_offset_snap_root_skip_disconnected 8
|
||||
#define _atom_offset_disconnected_snap_end 60
|
||||
#define _atom_offset_case_2_id_dispatch 8
|
||||
#define _atom_offset_pending_snap_end 50
|
||||
#define _atom_offset_id_dispatch_try_analog_stick 11
|
||||
#define _atom_offset_id_dispatch_snap_end 37
|
||||
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
||||
#define _atom_offset_analog_stick_snap_end 23
|
||||
#define _atom_offset_try_analog_pad_try_unsupported 11
|
||||
#define _atom_offset_analog_pad_snap_end 9
|
||||
|
||||
enum {
|
||||
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
|
||||
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
|
||||
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
|
||||
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
|
||||
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
|
||||
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
|
||||
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
|
||||
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
|
||||
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||
};
|
||||
|
||||
// --- atom: pad_apply_input (60 words) ---
|
||||
|
||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||
#define _atom_offset_dead_zone_skip_exit_stick 23
|
||||
#define _atom_offset_end_low_exit_stick 11
|
||||
|
||||
enum {
|
||||
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
|
||||
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
|
||||
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
|
||||
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
|
||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||
};
|
||||
|
||||
#pragma endregion hello_joypad.tape
|
||||
|
||||
@@ -0,0 +1,545 @@
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <assert.h>
|
||||
// #include "libgpu.h"
|
||||
// #include "libetc.h"
|
||||
// #include "libgte.h"
|
||||
|
||||
#include "duffle/dsl.h"
|
||||
#include "duffle/memory.h"
|
||||
#include "duffle/math.h"
|
||||
|
||||
#include "duffle/gcc_asm.h"
|
||||
#include "duffle/mips.h"
|
||||
#include "duffle/gp.h"
|
||||
#include "duffle/gte.h"
|
||||
#include "duffle/pad.h"
|
||||
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
#include "duffle/atom_dsl.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
#include "duffle/word_count.metadata.h"
|
||||
|
||||
#include "psyq.h"
|
||||
|
||||
# include "gen/hello_joypad.macs.h"
|
||||
# include "gen/hello_joypad.offsets.h"
|
||||
#include "hello_joypad.h"
|
||||
|
||||
#include "psyq.c"
|
||||
#include "hello_joypad.tape.c"
|
||||
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||
typedef Array_(PrimitiveBuffer, 2);
|
||||
typedef Struct_(PrimitiveArena) {
|
||||
A2_PrimitiveBuffer buf;
|
||||
U4 used;
|
||||
};
|
||||
|
||||
#define Cube_num_verts 8
|
||||
typedef Array_(V3_S2, Cube_num_verts);
|
||||
#define Cube_num_faces 6
|
||||
typedef Array_(V4_S2, Cube_num_faces);
|
||||
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||
{ -128, -128, -128 },
|
||||
{ 128, -128, -128 },
|
||||
{ 128, -128, 128 },
|
||||
{ -128, -128, 128 },
|
||||
{ -128, 128, -128 },
|
||||
{ 128, 128, -128 },
|
||||
{ 128, 128, 128 },
|
||||
{ -128, 128, 128 }
|
||||
};
|
||||
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||
{ 3, 2, 0, 1 },
|
||||
{ 0, 1, 4, 5 },
|
||||
{ 4, 5, 7, 6 },
|
||||
{ 1, 2, 5, 6 },
|
||||
{ 2, 3, 6, 7 },
|
||||
{ 3, 0, 7, 4 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||
return;
|
||||
}
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A8_V3_S2 verts;
|
||||
A6_V4_S2 faces;
|
||||
};
|
||||
|
||||
#define Floor_num_verts 4
|
||||
typedef Array_(V3_S2, Floor_num_verts);
|
||||
#define Floor_num_faces 2
|
||||
typedef Array_(V3_S2, Floor_num_faces);
|
||||
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||
{ -900, 0, -900 },
|
||||
{ -900, 0, 900 },
|
||||
{ 900, 0, -900 },
|
||||
{ 900, 0, 900 },
|
||||
};
|
||||
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||
{ 0, 1, 2 },
|
||||
{ 1, 3, 2 },
|
||||
};
|
||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||
};
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos;
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A4_V3_S2 verts;
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
|
||||
enum {
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
U4 MemTape[MemTape_Len];
|
||||
|
||||
DoubleBuffer screen_buf;
|
||||
A2_OrderingTable_Buffer ordering_tbl;
|
||||
PrimitiveArena primitives;
|
||||
S4 active_buf_id;
|
||||
|
||||
M3_S2 tform_world;
|
||||
|
||||
Ent_Cube cube;
|
||||
Ent_Floor floor;
|
||||
|
||||
PadBiosRaw pad_raw[2];
|
||||
PadState pad[2];
|
||||
|
||||
U4_V scratchpad; // d-cache
|
||||
};
|
||||
global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||
B1* next = buf + pa->used;
|
||||
pa->used += type_width;
|
||||
return next;
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
|
||||
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||
*
|
||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
|
||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||
*
|
||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
|
||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
{
|
||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||
(void)p0; (void)p1;
|
||||
|
||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||
// Use enums.
|
||||
|
||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||
* $a3 = 0x22 (immediate)
|
||||
* $t1 = 0x12 (function number)
|
||||
* $t2 = 0xB0 (BIOS B-table address) */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
||||
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
|
||||
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
|
||||
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
|
||||
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_rpins, r_use(p0), r_use(p1)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
|
||||
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||
u1_v(raw0)[0] = 0xFF;
|
||||
u1_v(raw1)[0] = 0xFF;
|
||||
|
||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
|
||||
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
}
|
||||
|
||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S4* active_buf_id)
|
||||
{
|
||||
reset_graph(0);
|
||||
|
||||
// Set the current initial buffer
|
||||
active_buf_id[0] = 0;
|
||||
|
||||
// Just setting env data, not interacting with console hw.
|
||||
// First buffer area
|
||||
displayenv_init(& r_(screen_buf->display)[0], 0, 0, ScreenRes_X, ScreenRes_Y);
|
||||
drawenv_init (& r_(screen_buf->draw )[0], 0, ScreenRes_Y, ScreenRes_X, ScreenRes_Y);
|
||||
// Second buffer area
|
||||
displayenv_init(& r_(screen_buf->display)[1], 0, ScreenRes_Y, ScreenRes_X, ScreenRes_Y);
|
||||
drawenv_init (& r_(screen_buf->draw )[1], 0, 0, ScreenRes_X, ScreenRes_Y);
|
||||
// Set the back/drawing buffer
|
||||
screen_buf->draw[0].enable_auto_clear = true;
|
||||
screen_buf->draw[1].enable_auto_clear = true;
|
||||
// Set the background clear color
|
||||
screen_buf->draw[0].initial_bg_color = rgb8( .r = 7, .g = 7, .b = 7 );
|
||||
screen_buf->draw[1].initial_bg_color = rgb8( .r = 7, .g = 7, .b = 7 );
|
||||
// screen_buf->draw[1].initial_bg_color = rgb8( .r = 47, .g = 13, .b = 0 );
|
||||
displayenv_put(& r_(screen_buf->display)[ active_buf_id[0] ]);
|
||||
drawenv_put (& r_(screen_buf->draw )[ active_buf_id[0] ]);
|
||||
|
||||
// Initialize and setup the GTE geometry offsets
|
||||
geom_init();
|
||||
geom_set_offset(ScreenRes_CenterX, ScreenRes_CenterY);
|
||||
geom_set_screen(ScreenZ);
|
||||
|
||||
set_display_enabled(1); // gp_DisplayEnabled
|
||||
}
|
||||
|
||||
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||
draw_sync(0);
|
||||
vsync(0);
|
||||
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||
{
|
||||
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||
pa->used = 0;
|
||||
}
|
||||
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||
}
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
if (0) // Pad Input (dead — kept for the source-as-written record; references the deleted `pad_state` field)
|
||||
{
|
||||
(void)Pad_Left; (void)Pad_Right; /* suppress unused-token warnings */
|
||||
if (false) {
|
||||
smem.cube.rot.y += 30;
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
if (false) {
|
||||
smem.cube.rot.y -= 30;
|
||||
smem.floor.rot.y -= 5;
|
||||
}
|
||||
}
|
||||
if (1) // Pad Input (Tape version)
|
||||
{
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
/* BIOS-owned polling: per-frame snapshot of both ports. */
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[0]);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[1]);
|
||||
tb_data_(state, & smem.pad[1]);
|
||||
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
|
||||
tb_emit_(pad_apply_input);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_data_(cube_rot, & smem.cube.rot);
|
||||
tb_data_(floor_rot, & smem.floor.rot);
|
||||
}
|
||||
}
|
||||
|
||||
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||
|
||||
// Update the position based on acceleration and velocity
|
||||
gknown V3_S4_R pos = & smem.cube.pos;
|
||||
gknown V3_S4_R vel = & smem.cube.vel;
|
||||
gknown V3_S4_R acc = & smem.cube.accel;
|
||||
add_v3s4(vel, acc[0]);
|
||||
add_v3s4_fp(pos, vel[0]);
|
||||
// vel->x += acc->x;
|
||||
// vel->y += acc->y;
|
||||
// vel->z += acc->z;
|
||||
// pos->x += vel->x;
|
||||
// pos->y += vel->y;
|
||||
// pos->z += vel->z;
|
||||
|
||||
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||
|
||||
// Prep
|
||||
S4 nclip = 0;
|
||||
S4 orderingtbl_z = 0;
|
||||
A2_S2 p; //???
|
||||
S4 flag; //????
|
||||
|
||||
|
||||
// Draw Cube
|
||||
if (0)
|
||||
{
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
// gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
for (U4 face_id = 0; face_id < Cube_num_faces; face_id += 1)
|
||||
{
|
||||
Poly_G4* quad = prim_alloc(Poly_G4); set_poly_g4(quad);
|
||||
quad->c0 = rgb8(255, 0, 255);
|
||||
quad->c1 = rgb8(255, 255, 0);
|
||||
quad->c2 = rgb8( 0, 255, 255);
|
||||
quad->c3 = rgb8( 0, 255, 0);
|
||||
|
||||
V4_S2* face = & smem.cube.faces[face_id];
|
||||
V3_S2* p0 = & smem.cube.verts[face->x];
|
||||
V3_S2* p1 = & smem.cube.verts[face->y];
|
||||
V3_S2* p2 = & smem.cube.verts[face->z];
|
||||
V3_S2* p3 = & smem.cube.verts[face->w];
|
||||
|
||||
nclip = rtp_avg_nclip_a4_v3s2(
|
||||
p0, p1, p2, p3,
|
||||
& quad->p0, & quad->p1, & quad->p2, & quad->p3,
|
||||
& p, & orderingtbl_z, & flag
|
||||
);
|
||||
if (nclip <= 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], quad);
|
||||
}
|
||||
}
|
||||
// smem.cube.rot.x += 6;
|
||||
// smem.cube.rot.y += 8;
|
||||
// smem.cube.rot.z += 12;
|
||||
smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw cube (tape method) - two triangles per face
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, rbind_cube_g4_face);
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.cube.faces));
|
||||
tb_data(& tb, u4_(smem.cube.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
|
||||
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||
tb_emit(& tb, cube_g4_face);
|
||||
}
|
||||
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));
|
||||
|
||||
// smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw Floor
|
||||
if (0)
|
||||
{
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
for (U4 face_id = 0; face_id < Floor_num_faces; face_id += 1)
|
||||
{
|
||||
Poly_F3* tri = prim_alloc(Poly_F3); set_poly_f3(tri);
|
||||
tri->color = rgb8(255, 255, 255);
|
||||
|
||||
V3_S2* face = & smem.floor.faces[face_id];
|
||||
register V3_S2* p0 rgcc(R_T4) = & smem.floor.verts[face->x];
|
||||
register V3_S2* p1 rgcc(R_T5) = & smem.floor.verts[face->y];
|
||||
register V3_S2* p2 rgcc(R_T6) = & smem.floor.verts[face->z];
|
||||
|
||||
gte_load_v0(p0, R_T4);
|
||||
/*
|
||||
asm volatile( ".word " "%0" ", %1" : :
|
||||
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_xy & REG_MASK) << RT_SHIFT) | (0 & IMM_MASK)),
|
||||
"i"(((op_lwc2 & OPCODE_MASK) << OPCODE_SHIFT) | ((R_T4 & REG_MASK) << RS_SHIFT) | ((gte_in_v0_z & REG_MASK) << RT_SHIFT) | (GTE_Z_Offset & IMM_MASK)),
|
||||
"r"(p0) :
|
||||
"$2", "$8", "$9", "$31", "memory"
|
||||
);
|
||||
*/
|
||||
gte_load_v1(p1, R_T5);
|
||||
gte_load_v2(p2, R_T6);
|
||||
|
||||
gte_rtpt();
|
||||
gte_nclip();
|
||||
gte_stotz(& nclip);
|
||||
|
||||
// nclip = rtp_avg_nclip_a3_v3s2(p0, p1, p2
|
||||
// , & tri->p0, & tri->p1, & tri->p2
|
||||
// , & p, & orderingtbl_z, & flag
|
||||
// );
|
||||
// if (nclip <= 0) {
|
||||
// continue;
|
||||
// }
|
||||
|
||||
if (nclip > 0 ) {
|
||||
gte_stsxy3(& tri->p0, & tri->p1, & tri->p2);
|
||||
gte_avsz3();
|
||||
gte_stotz(& orderingtbl_z);
|
||||
|
||||
if ((orderingtbl_z > 0) && (orderingtbl_z < OrderingTbl_Len)) {
|
||||
orderingtbl_add_primitive(ordering_buf[orderingtbl_z], tri);
|
||||
}
|
||||
}
|
||||
}
|
||||
smem.floor.rot.y += 5;
|
||||
}
|
||||
// Draw floor tape method
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
|
||||
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||
// The tape atoms in-flight should not need to care.
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, set_gte_world);
|
||||
tb_data(& tb, u4_(& smem.tform_world));
|
||||
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
tb_data(& tb, u4_(ordering_buf));
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
tb_emit(& tb, floor_f3_face);
|
||||
}
|
||||
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||
tb_emit(& tb, sync_primitive_arena);
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape.
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
// smem.floor.rot.y += 5;
|
||||
}
|
||||
// --- TAPE DIAGNOSTICS ---
|
||||
if (0)
|
||||
{
|
||||
LP_ U4 mem_temp_tape[512]; FArena tape_arena; farena_init(& tape_arena, slice_ut_arr(mem_temp_tape));
|
||||
TapeBuilder tb = tb_make_old(& tape_arena); tb_scope(& tb) {
|
||||
// Skip set_gte_world atom for diagnostics to isolate the triangle loop
|
||||
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||
// tb_emit(& tb, code_diag_yield);
|
||||
// tb_emit(& tb, code_diag_color);
|
||||
// tb_emit(& tb, code_diag_gte);
|
||||
}
|
||||
}
|
||||
B1* prim_cursor = (B1*)r_(pa->buf)[smem.active_buf_id] + pa->used;
|
||||
tape_run(tb_slice(tb));
|
||||
pa->used = (U4)prim_cursor - (U4)r_(pa->buf)[smem.active_buf_id];
|
||||
}
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
void render(void) {
|
||||
}
|
||||
|
||||
int main(void)
|
||||
{
|
||||
smem = (SMemory){0};
|
||||
smem.scratchpad = C_(U4_V, 0x1F800000);
|
||||
// smem.primitives.used = 0;
|
||||
// smem.active_buf_id = 0;
|
||||
/*Persistent Entity Setup*/{
|
||||
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||
Ent_Cube* cube = & smem.cube;
|
||||
cube->rot = v3s2(0, 0, 0);
|
||||
cube->scale = v3s4_fp_one();
|
||||
cube->accel = v3s4(0, 1, 0);
|
||||
cube->pos = v3s4(0, -400, 1800);
|
||||
}
|
||||
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||
Ent_Floor* floor = & smem.floor;
|
||||
floor->rot = v3s2(0, 0, 0);
|
||||
floor->pos = v3s4(0, 450, 1800);
|
||||
floor->scale = v3s4_fp_one();
|
||||
}
|
||||
}
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
|
||||
reset_graph(0);
|
||||
/* Direct BIOS: poll both ports during VBlank. */
|
||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||
/* Pinned registers for the GPU init atom. */
|
||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
tb_emit(& tb, screen_env_init);
|
||||
tb_emit(& tb, gp_screen_init);
|
||||
}
|
||||
}
|
||||
while (1) {
|
||||
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
update(pa, ordering_buf);
|
||||
render();
|
||||
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||
};
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "duffle/dsl.h"
|
||||
# include "duffle/math.h"
|
||||
# include "duffle/gp.h"
|
||||
# include "duffle/pad.h"
|
||||
#endif
|
||||
|
||||
enum {
|
||||
PrimitiveBuff_Len = 4096,
|
||||
OrderingTbl_Len = 2048
|
||||
};
|
||||
|
||||
enum {
|
||||
ScreenRes_X = 320,
|
||||
ScreenRes_Y = 240,
|
||||
ScreenZ = 320,
|
||||
ScreenRes_CenterX = (ScreenRes_X >> 1),
|
||||
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||
};
|
||||
|
||||
enum {
|
||||
fp_one = (1 << 12),
|
||||
};
|
||||
|
||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||
@@ -0,0 +1,705 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "duffle/gen/duffle.macs.h"
|
||||
# include "duffle/gen/duffle.offsets.h"
|
||||
# include "duffle/atom_dsl.h"
|
||||
# include "duffle/lottes_tape.h"
|
||||
# include "duffle/mips.h"
|
||||
# include "duffle/gte.h"
|
||||
# include "duffle/gp.h"
|
||||
# include "duffle/pad.h"
|
||||
# include "duffle/word_count.metadata.h"
|
||||
# include "psyq.h"
|
||||
# include "gen/hello_joypad.offsets.h"
|
||||
# include "gen/hello_joypad.macs.h"
|
||||
# include "hello_joypad.h"
|
||||
#endif
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v2s2(U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, {
|
||||
load_half( rs_x, r_base, O_(V3_S2,x)),
|
||||
load_half( rs_y, r_base, O_(V3_S2,y)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, {
|
||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, {
|
||||
store_byte(rr, base, offset + O_(DrawEnv,initial_bg_color.r)),
|
||||
store_byte(rg, base, offset + O_(DrawEnv,initial_bg_color.g)),
|
||||
store_byte(rb, base, offset + O_(DrawEnv,initial_bg_color.b)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_gcmd_push(U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_gcmd_push, {
|
||||
load_upper_i(reg_transfer, cmd >> 16),
|
||||
or_i_self( reg_transfer, cmd & 0xFFFF),
|
||||
store_word( reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_draw_env, {
|
||||
/*
|
||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||
* References:
|
||||
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
|
||||
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
|
||||
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
|
||||
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
|
||||
*
|
||||
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
|
||||
* ./toolchain/psyq-4_7/lib/libgpu.a
|
||||
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||
*
|
||||
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||
*/
|
||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
|
||||
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
|
||||
|
||||
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
|
||||
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
|
||||
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
|
||||
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
|
||||
|
||||
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
|
||||
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||
|
||||
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
/* DIAGNOSTIC 1: Pure tape loop test */
|
||||
internal MipsAtom_(diag_yield) { mac_yield() };
|
||||
|
||||
/* DIAGNOSTIC 2: Pure memory test (No GTE). Draws a fixed cyan triangle. */
|
||||
internal MipsAtom_(diag_color) {
|
||||
store_word( R_0, R_T7, 0),
|
||||
load_upper_i(R_AT, gp0_cmd_poly_f3 << 8 | 0xFF), /* High: MipsCode Poly_F3(0x20) + Color B:FF */
|
||||
or_i_self( R_AT, 0xFF00), /* Low: Color G:FF, R:00 (Cyan) */
|
||||
store_word( R_AT, R_T7, 4),
|
||||
|
||||
/* Fake coordinates - Swapped winding order to prevent GPU culling! */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 8), /* (16, 16) */
|
||||
load_upper_i(R_AT, 0x0050), or_i_self(R_AT, 0x0010), store_word(R_AT, R_T7, 12), /* (80, 16) */
|
||||
load_upper_i(R_AT, 0x0010), or_i_self(R_AT, 0x0050), store_word(R_AT, R_T7, 16), /* (16, 80) */
|
||||
|
||||
add_ui( R_T1, R_0, 10),
|
||||
shift_lleft_self(R_T1, S_(U4)/2),
|
||||
add_u_self( R_T1, R_T6),
|
||||
|
||||
load_word( R_AT, R_T1, 0),
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
store_word( R_AT, R_T7, 0),
|
||||
shift_lleft(R_AT, R_T7, S_(PolyTag_len_bits)), shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
or_u_self( R_AT, R_V0),
|
||||
store_word( R_AT, R_T1, 0),
|
||||
|
||||
add_ui(R_T7, R_T7, 20),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* DIAGNOSTIC 3: Pure GTE test (No Memory Writes) */
|
||||
internal MipsAtom_(diag_gte) {
|
||||
/* Load 3 indices */
|
||||
load_half_u(R_T0, R_T4, 0),
|
||||
load_half_u(R_T1, R_T4, 2),
|
||||
load_half_u(R_T2, R_T4, 4),
|
||||
|
||||
/* Load Vertices into GTE */
|
||||
shift_lleft( R_AT, R_T0, 3), add_u( R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
shift_lleft( R_AT, R_T1, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
|
||||
shift_lleft(R_AT, R_T2, 3), add_u(R_AT, R_AT, R_T5),
|
||||
load_word(R_V0, R_AT, 0), load_word(R_V1, R_AT, 4),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
|
||||
/* Run Math */
|
||||
nop2, gte_cmdw_rtpt,
|
||||
nop2, gte_cmdw_nclip,
|
||||
nop2,
|
||||
|
||||
/* Advance Face Cursor and Yield */
|
||||
add_ui(R_T4, R_T4, 8),
|
||||
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
enum {
|
||||
R_ScreenX = R_T5 atom_reg atom_type(U2),
|
||||
R_ScreenY = R_T6 atom_reg atom_type(U2),
|
||||
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
|
||||
#define R_ScreenBuf_Code R_T7_Code
|
||||
};
|
||||
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||
) {
|
||||
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
|
||||
|
||||
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
|
||||
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
|
||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||
add_ui(R_T0, R_0, 1),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
|
||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||
add_ui(R_T0, R_0, 7),
|
||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
enum {
|
||||
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||
#define R_IO_BaseAddr_Code R_T4_Code
|
||||
};
|
||||
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO */
|
||||
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ */
|
||||
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON */
|
||||
mac_gcmd_push(gp1_word_dma_to_gpu(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU→GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. */
|
||||
mac_gcmd_push(gp1_word_StartDisplayArea(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0) */
|
||||
|
||||
/* GP1: DisplayMode + Display Ranges */
|
||||
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
|
||||
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||
load_upper_i(R_T5, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_T5, gte_cr_OFX_Code),
|
||||
load_upper_i(R_T5, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_T5, gte_cr_OFY_Code),
|
||||
|
||||
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||
add_ui(R_T5, R_0, ScreenZ), gte_mv_to_ctrl_r(R_T5, gte_cr_H_Code),
|
||||
|
||||
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||
mac_gcmd_push(gp1_word_DisplayOn(), R_T5, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_CubeTri) {
|
||||
U4 PrimCursor;
|
||||
V4_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
){
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), nop,
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
mac_gte_store_g4_p012(),
|
||||
gte_cmdw_rotate_translate_perspective_single,
|
||||
mac_gte_store_g4_p3(),
|
||||
|
||||
gte_cmdw_avg_sort_z4,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(),
|
||||
mac_format_g4_color(
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(cull)
|
||||
|
||||
atom_label(cube_g4_face_exit)
|
||||
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_FloorTri) {
|
||||
U4 PrimCursor;
|
||||
V3_S2* FaceCursor;
|
||||
V3_S2* VertBase;
|
||||
U4* OtBase;
|
||||
};
|
||||
internal
|
||||
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||
, atom_reads(R_TapePtr)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||
){
|
||||
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// atom_dbg_skip
|
||||
internal
|
||||
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
/* Culling (Branch forward if Backface) */
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||
/* Format Primitive */
|
||||
mac_gte_store_f3(),
|
||||
|
||||
/* Calculate Depth */
|
||||
gte_avg_sort_z3,
|
||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
// end: branch(bounds_chk)
|
||||
// end: branch(culling)
|
||||
|
||||
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||
atom_label(floor_f3_face_exit)
|
||||
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
/* ----- pad_bios_snapshot -----
|
||||
* Per-frame snapshot of one BIOS pad buffer into PadState.
|
||||
* Decoder (branch ladder on raw[0] status + raw[1] id):
|
||||
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
|
||||
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
|
||||
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
|
||||
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
|
||||
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
|
||||
* 6. else -> Unsupported (buttons=0, axes=0x80)
|
||||
*
|
||||
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
|
||||
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
|
||||
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||
*
|
||||
* Register use (atom-local; no wave-context touched):
|
||||
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
|
||||
* R_T1 = state base (kept throughout; all stores go through R_T1)
|
||||
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
|
||||
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
|
||||
* R_T4 = scratch (shifts, compares, immediate loads, store values)
|
||||
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
|
||||
*/
|
||||
enum {
|
||||
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||
R_PadState = R_T1 atom_reg,
|
||||
R_RawStatus = R_T2 atom_reg,
|
||||
R_RawId = R_T3 atom_reg,
|
||||
};
|
||||
typedef Struct_(Binds_PadBiosSnapshot) {
|
||||
PadBiosRaw* raw;
|
||||
PadState* state;
|
||||
};
|
||||
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||
) {
|
||||
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||
|
||||
/* === Read raw[0] (status) + raw[1] (id) */
|
||||
load_byte_u(R_RawStatus, R_PadRaw, 0),
|
||||
load_byte_u(R_RawId, R_PadRaw, 1),
|
||||
|
||||
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||
|
||||
atom_label(disconnected) /* === Disconnected body. */
|
||||
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
branch_equal(R_0, R_0, atom_offset(disconnected, snap_end)), nop,
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(disconnected, snap_end)), nop,
|
||||
atom_label(skip_disconnected)
|
||||
|
||||
/* === Case 2: Pending (status == 0 && id == 0)
|
||||
* Combined check: if (status | id) != 0 then skip to id_dispatch.
|
||||
* Falls through to the Pending case only when both are zero. */
|
||||
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
|
||||
|
||||
atom_label(pending) /* === Pending body */
|
||||
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
branch_equal(R_0, R_0, atom_offset(pending, snap_end)), nop,
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(pending, snap_end)), nop,
|
||||
|
||||
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||
|
||||
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
|
||||
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
|
||||
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
|
||||
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
|
||||
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
|
||||
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
store_word( R_T5, R_PadState, O_(PadState,left_x)),
|
||||
add_ui( R_T4, R_0, 0x41),
|
||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||
|
||||
branch_equal(R_0, R_0, atom_offset(id_dispatch, snap_end)), nop,
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(id_dispatch, snap_end)), nop,
|
||||
|
||||
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||
|
||||
atom_label(analog_stick) /* === AnalogStick body
|
||||
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
|
||||
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
|
||||
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
|
||||
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
|
||||
store_byte( R_T5, R_PadState, O_(PadState,id)),
|
||||
branch_equal(R_0, R_0, atom_offset(analog_stick, snap_end)), nop,
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(analog_stick, snap_end)), nop,
|
||||
|
||||
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||
and_i( R_T4, R_RawId, 0xF0),
|
||||
add_ui( R_T5, R_0, 0x70),
|
||||
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||
|
||||
atom_label(analog_pad) /* === AnalogPad body
|
||||
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
|
||||
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
|
||||
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
|
||||
branch_equal(R_0, R_0, atom_offset(analog_pad, snap_end)), nop,
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(analog_pad, snap_end)), nop,
|
||||
|
||||
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
|
||||
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
|
||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||
/* Fall through to snap_end. */
|
||||
|
||||
atom_label(snap_end)
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
/* ----- pad_apply_input -----
|
||||
* Reads pad[0].buttons + pad[0].left_x;
|
||||
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
|
||||
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
|
||||
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
|
||||
* - Analog stick X (dead zone 0x70..0x90):
|
||||
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
|
||||
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
|
||||
* - D-pad + analog deltas add when used together.
|
||||
*
|
||||
* Convention:
|
||||
* pad_state = 0 means no buttons active.
|
||||
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
|
||||
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
|
||||
*
|
||||
* Signed-delta trick:
|
||||
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
|
||||
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
|
||||
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
|
||||
*/
|
||||
typedef Struct_(Binds_PadApplyInput) {
|
||||
PadState* state;
|
||||
V3_S2* cube_rot;
|
||||
V3_S2* floor_rot;
|
||||
};
|
||||
enum {
|
||||
R_PadStateT5 = R_T5 atom_reg,
|
||||
R_CubeRot = R_T1 atom_reg,
|
||||
R_FloorRot = R_T2 atom_reg,
|
||||
};
|
||||
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||
, atom_writes( R_CubeRot, R_FloorRot)
|
||||
) {
|
||||
/* Pop Binds from tape (state, cube_rot, floor_rot) */
|
||||
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0. */
|
||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
||||
// Note(Ed): Potential op with delay slot?
|
||||
|
||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, 30),
|
||||
add_si( R_T3, R_T3, 5),
|
||||
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
atom_label(exit_dpad_left)
|
||||
|
||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, -30),
|
||||
add_si( R_T3, R_T3, -5),
|
||||
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
atom_label(exit_dpad_right)
|
||||
|
||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
|
||||
|
||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||
|
||||
atom_label(dead_check_upper)
|
||||
/* left_x >= 0x70 → check upper bound. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
|
||||
add_ui( R_T4, R_0, 0x90),
|
||||
|
||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||
branch_equal(R_0, R_0, atom_offset(dead_zone_skip, exit_stick)), nop,
|
||||
/* Fall-through = left_x in [0x70, 0x90] (dead zone); skip analog entirely. */
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(dead_zone_skip, exit_stick)), nop,
|
||||
|
||||
atom_label(dead_low_active)
|
||||
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
|
||||
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
|
||||
/* delta = 0x80 - left_x (positive). */
|
||||
|
||||
/* R_T4 = cube_delta */
|
||||
shift_aright(R_T4, R_T3, 2),
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
branch_equal(R_0, R_0, atom_offset(end_low, exit_stick)), nop,
|
||||
// TODO(Ed): Lua metaprogram: Support jump instruction here..
|
||||
// jump(atom_offset(end_low, exit_stick)), nop,
|
||||
|
||||
atom_label(dead_high_active)
|
||||
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
|
||||
sub_u( R_T3, R_T4, R_T3),
|
||||
/* delta = 0x80 - left_x (signed negative). */
|
||||
|
||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
|
||||
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
atom_label(exit_stick)
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
#pragma endregion Baked Atoms
|
||||
@@ -0,0 +1,3 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "psyq.h"
|
||||
#endif
|
||||
@@ -0,0 +1,103 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "duffle/dsl.h"
|
||||
# include "duffle/math.h"
|
||||
# include "duffle/gp.h"
|
||||
#endif
|
||||
|
||||
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||
typedef Struct_(DrawEnv) {
|
||||
Rect_S2 clip_area;
|
||||
V2_S2 drawing_offset[2];
|
||||
Rect_S2 texture_window;
|
||||
S2 texture_page;
|
||||
B1 flag_dither;
|
||||
B1 flag_draw_on_display;
|
||||
B1 enable_auto_clear;
|
||||
RGB8 initial_bg_color;
|
||||
DrawEnv_Packed dr_env; // reserved
|
||||
};
|
||||
typedef Struct_(DisplayEnv) {
|
||||
Rect_S2 display_area;
|
||||
Rect_S2 screen;
|
||||
B1 vinterlace;
|
||||
B1 color24;
|
||||
B1 pad0;
|
||||
B1 pad1;
|
||||
};
|
||||
typedef Array_(DrawEnv, 2);
|
||||
typedef Array_(DisplayEnv, 2);
|
||||
|
||||
typedef Struct_(DoubleBuffer) {
|
||||
A2_DrawEnv draw;
|
||||
A2_DisplayEnv display;
|
||||
};
|
||||
|
||||
DisplayEnv* displayenv_init(DisplayEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDispEnv");
|
||||
DrawEnv* drawenv_init (DrawEnv* env, S4 x, S4 y, S4 w, S4 h) asm("SetDefDrawEnv");
|
||||
|
||||
DisplayEnv* displayenv_put(DisplayEnv* env) asm("PutDispEnv");
|
||||
DrawEnv* drawenv_put (DrawEnv* env) asm("PutDrawEnv");
|
||||
|
||||
U4 geom_init(void) asm("InitGeom");
|
||||
void geom_set_offset(U4 x, U4 y) asm("SetGeomOffset");
|
||||
void geom_set_screen(U4 h) asm("SetGeomScreen");
|
||||
|
||||
U4* orderingtbl_clear_reverse(U4* ot, U4 len) asm("ClearOTagR");
|
||||
|
||||
U4 reset_graph(U4 mode) asm("ResetGraph");
|
||||
void set_display_enabled(U4 mask) asm("SetDispMask");
|
||||
|
||||
U4 draw_sync(U4 mode) asm("DrawSync");
|
||||
U4 vsync(U4 mode) asm("VSync");
|
||||
|
||||
void draw_orderingtbl(U4* buf) asm("DrawOTag");
|
||||
|
||||
typedef Struct_(Tile) {
|
||||
U4 tag;
|
||||
RGB8 color;
|
||||
B1 code;
|
||||
Rect_S2 rect;
|
||||
};
|
||||
|
||||
/*
|
||||
Linear Algebra
|
||||
*/
|
||||
|
||||
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix");
|
||||
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix");
|
||||
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||
|
||||
// Rotation, Translation, Perspective
|
||||
|
||||
S4 rtp_v3s2_raw(V3_S2* vec, S4* xy, S4* pp, S4* flag) asm("RotTransPers");
|
||||
FI_ S4 rtp_v3s2(V3_S2* vec, V2_S2* xy, A2_S2* pp, S4* flag) { return rtp_v3s2_raw(vec, C_(S4*R_, & xy->x), C_(S4*R_, pp), r_(flag)); }
|
||||
|
||||
S4 rtp_avg_nclip_a3_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, S4* xy1, S4* xy2, S4* xy3, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip3");
|
||||
FI_ S4 rtp_avg_nclip_a3_v3s2(
|
||||
V3_S2* v0, V3_S2* v1, V3_S2* v2,
|
||||
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2,
|
||||
A2_S2* pp, S4* otz, S4* flag
|
||||
){
|
||||
return rtp_avg_nclip_a3_v3s2_raw(
|
||||
v0, v1, v2,
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
S4 rtp_avg_nclip_a4_v3s2_raw(V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3, S4* xy1, S4* xy2, S4* xy3, S4* xy4, S4* pp, S4* otz, S4* flag) asm("RotAverageNclip4");
|
||||
FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||
V3_S2* v0, V3_S2* v1, V3_S2* v2, V3_S2* v3,
|
||||
V2_S2* xy0, V2_S2* xy1, V2_S2* xy2, V2_S2* xy3,
|
||||
A2_S2* pp, S4* otz, S4* flag
|
||||
){
|
||||
return rtp_avg_nclip_a4_v3s2_raw(
|
||||
v0, v1, v2, v3,
|
||||
C_(S4*R_, xy0), C_(S4*R_, xy1), C_(S4*R_, xy2), C_(S4*R_, xy3),
|
||||
C_(S4*R_, pp), C_(S4*R_, otz), C_(S4*R_, flag)
|
||||
);
|
||||
}
|
||||
|
||||
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix");
|
||||
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix");
|
||||
@@ -0,0 +1,659 @@
|
||||
|
||||
#if 0 /* ac_pad_sio_write_pad_state — superseded by pad_bios_snapshot */
|
||||
|
||||
/* ============================================================
|
||||
* raw_sio_pad_poll_20260802 — superseded by bios_pad_buffer_snapshot_20260803.
|
||||
* The doomed raw-SIO production atoms (ac_pad_sio_write_pad_state,
|
||||
* pad_sio_init, pad_sio_step, pad_sio_diag_pin, pad_sio_diag_byte_exchange)
|
||||
* reference symbols that were removed from code/duffle/pad.h during
|
||||
* Phase 1. Each is wrapped in a narrow `#if 0` so the C compile skips
|
||||
* the body while the source-as-written text stays in place for the
|
||||
* Phase 5.1 deletion pass. The wrap is removed (and the bodies are
|
||||
* deleted) by Phase 5.1 of this track.
|
||||
* ============================================================ */
|
||||
|
||||
* Writes the per-port PadState in 5 instructions plus 4 store_word calls (status,
|
||||
* buttons, left_x/y/right_x/right_y packed, attempt). The provisional decode publishes
|
||||
* 0x0000FFFF buttons + centered axes on every path until response-byte decode lands.
|
||||
*
|
||||
* Args:
|
||||
* status_val - the PadSioStatus enum value to publish
|
||||
* state_ptr_reg - the PadState* base (R_PadState at the call site)
|
||||
* scratch_reg - scratch register for the value being stored (e.g., R_T0)
|
||||
*
|
||||
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
||||
* two-instruction zero-extended buttons load).
|
||||
*/
|
||||
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, {
|
||||
add_ui(scratch_reg, R_0, status_val),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
||||
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
||||
* libetc convention. Build it with LUI + ORI so addiu does not
|
||||
* sign-extend 0xFFFF to 0xFFFFFFFF. */
|
||||
load_upper_i(scratch_reg, 0x0000),
|
||||
or_i(scratch_reg, scratch_reg, 0xFFFF),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,buttons)),
|
||||
add_ui(scratch_reg, R_0, 0x80808080),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,left_x)),
|
||||
add_ui(scratch_reg, R_0, 0),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,attempt))
|
||||
})
|
||||
#endif /* end ac_pad_sio_write_pad_state wrap */
|
||||
|
||||
/* ----- pad_sio_init -----
|
||||
* Boot-time SIO0 init. Caller pins R_T6 = sio_base_addr0.
|
||||
* Issues SIO CTRL=0x0040 (reset), MODE=0x000D, BAUD=0x0088.
|
||||
* (Phase 2 fills the body.)
|
||||
*/
|
||||
#if 0 /* pad_sio_init — superseded by pad_bios_init_start (Phase 1.3) */
|
||||
internal MipsAtom_(pad_sio_init) atom_info(atom_phase(pad_init)
|
||||
, atom_reads(R_T5, R_T6)
|
||||
, atom_writes(R_T5, R_T6)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly load the KSEG1 base into R_T6 at the top of
|
||||
* the atom body. The rgcc(R_PadSioBase) binding in main() pins R_T6 = base
|
||||
* when main() runs, but $12 is caller-saved per the O32 ABI — when tape_run
|
||||
* is invoked, R_T6 is fair game. The atom body cannot rely on the value. */
|
||||
load_upper_i(R_T6, pad_IO_KSEG1_BASE >> 16), /* R_T6 high 16 = 0xBF80 */
|
||||
or_i(R_T6, R_T6, pad_IO_KSEG1_BASE & 0xFFFF), /* R_T6 = 0xBF800000 */
|
||||
|
||||
/* SIO CTRL = 0x0040 (reset) */
|
||||
add_ui(R_T5, R_0, pad_SIO_CTRL_RESET),
|
||||
store_half(R_T5, R_T6, pad_SIO_CTRL_OFFSET),
|
||||
/* SIO MODE = 0x000D (MUL1, 8-bit, no parity, idle-high) */
|
||||
add_ui(R_T5, R_0, pad_SIO_MODE_INIT),
|
||||
store_half(R_T5, R_T6, pad_SIO_MODE_OFFSET),
|
||||
/* SIO BAUD = 0x0088 (~250 kHz) */
|
||||
add_ui(R_T5, R_0, pad_SIO_BAUD_INIT),
|
||||
store_half(R_T5, R_T6, pad_SIO_BAUD_OFFSET),
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_init wrap */
|
||||
|
||||
/* ----- pad_sio_step -----
|
||||
* Per-frame bounded raw-SIO transaction. Reads PadState pointers + SIO
|
||||
* base addresses from Binds_PadSioStep; writes per-port status +
|
||||
* buttons + axes into smem.pad[0..1].
|
||||
* Body shape (per spec §"Transaction model (per port, per pad_sio_step)"):
|
||||
* port 0: CTRL=CLEANUP → settle → CTRL=port-select → settle → exchange 5
|
||||
* bytes (addr + 0x42 0x00 0x00 0x00) → decode → write PadState[0]
|
||||
* → CTRL=CLEANUP.
|
||||
* port 1: swap scratch regs (sio_base_addr1 → R_PadSioBase, state1 →
|
||||
* R_PadState) → mirror port 0 sequence.
|
||||
*
|
||||
* Bounded-loop semantics: every countdown is wrapped in
|
||||
* add_ui_self(R_T1, -1) + branch_ne(R_T1, R_0, ...)
|
||||
* with a known maximum (pad_SIO_SETTLE_BEFORE_TX=1000, pad_SIO_SETTLE_AFTER_TX=2000,
|
||||
* pad_SIO_WAIT_BUDGET=4096). The static-analysis pass currently reports
|
||||
* has_loops = true; the follow-up metaprogram track that learns modeled-bounded
|
||||
* loops is out of scope here (per spec §"Risks").
|
||||
*
|
||||
* Scratch register strategy:
|
||||
* R_PadStatus = R_T4 — RESERVED for port-1 swap (holds state1)
|
||||
* R_PadCountdown = R_T5 — RESERVED for port-1 swap (holds sio_base_addr1)
|
||||
* R_T0 — byte-exchange value + STAT read (clobbered freely)
|
||||
* R_T1 — countdown budget (clobbered freely)
|
||||
* R_PadState = R_T7 — PadState* (preserved for PadState writes)
|
||||
* R_PadSioBase = R_T6 — SIO base (preserved through the port)
|
||||
*
|
||||
* Response decode (Task 3.1 teaching scope):
|
||||
* - status = PadSioStatus_Digital (hardcoded)
|
||||
* - buttons = 0xFFFF (no buttons pressed in the provisional libetc
|
||||
* convention; full response-byte decode is follow-up)
|
||||
* - axes = 0x80808080 (centered: left_x=0x80, left_y=0x80,
|
||||
* right_x=0x80, right_y=0x80)
|
||||
* - attempt = 0
|
||||
* - DualShock handshake (0x43 0x01 → 0x44 0x01 0x03 → 0x43 0x00) is
|
||||
* follow-up scope; the hardcoded digital decode is a placeholder.
|
||||
*
|
||||
* Both ports raise /CS (CTRL = pad_SIO_CTRL_CLEANUP) before exit. Both ports
|
||||
* treat response timeout as PadSioStatus_Disconnected per the spec §"Failure
|
||||
* handling" + the canonical per-port timeout semantics.
|
||||
*/
|
||||
#if 0 /* pad_sio_step — superseded by pad_bios_snapshot (Phase 2.1) */
|
||||
internal MipsAtom_(pad_sio_step) atom_info(atom_bind(Binds_PadSioStep)
|
||||
, atom_reads(R_TapePtr, R_PadSioBase, R_PadState, R_PadStatus, R_PadCountdown)
|
||||
, atom_writes(R_PadStatus, R_PadCountdown)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
|
||||
* top. The rgcc() binding in main() does NOT survive the tape_run call
|
||||
* because R_T6 is caller-saved per the O32 ABI. The pad_sio_init atom
|
||||
* (also in the per-frame tape) reloads R_T6 separately. */
|
||||
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
|
||||
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
|
||||
|
||||
/* Pop Binds from tape (in Binds_PadSioStep declaration order) */
|
||||
load_word(R_PadState, R_TapePtr, O_(Binds_PadSioStep,state0)),
|
||||
load_word(R_PadStatus, R_TapePtr, O_(Binds_PadSioStep,state1)), /* reserved for port-1 swap */
|
||||
load_word(R_PadSioBase, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr0)),
|
||||
load_word(R_PadCountdown, R_TapePtr, O_(Binds_PadSioStep,sio_base_addr1)), /* reserved for port-1 swap */
|
||||
add_ui_self(R_TapePtr, S_(Binds_PadSioStep)),
|
||||
|
||||
/* ============== PORT 0 TRANSACTION ============== */
|
||||
/* Use R_T0 (byte value / STAT read) + R_T1 (countdown) as scratch.
|
||||
* R_PadStatus (state1) + R_PadCountdown (sio_base_addr1) are preserved
|
||||
* through the port-0 body and swapped into R_PadSioBase + R_PadState
|
||||
* at atom_offset(port1_start, ...) below. */
|
||||
|
||||
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
|
||||
atom_label(settle_pre_port0)
|
||||
nop, /* BD slot */
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_pre_port0, settle_pre_port0)),
|
||||
|
||||
/* 2. Port-select: CTRL = 0x0003 (TX enable + DTR /CS) for port 0 */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS), /* set /CS line low */
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
|
||||
atom_label(settle_post_port0)
|
||||
nop,
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_post_port0, settle_post_port0)),
|
||||
|
||||
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
|
||||
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack0_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port0, ack0_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack0_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port0, wait_ack0_port0)),
|
||||
/* RX timeout → mark disconnected; skip to port 1 */
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack0, port1_start)),
|
||||
|
||||
atom_label(ack0_received_port0)
|
||||
/* Read open-bus response byte 0 — discard per docs/psx-spx §controllersandmemorycards.md */
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Confirm RX FIFO drained before sending byte 1. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel0_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port0, ack_released_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel0_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port0, wait_ackrel0_port0)),
|
||||
/* RX-drain timeout → disconnected; skip to port 1 */
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel0, port1_start)),
|
||||
|
||||
atom_label(ack_released_port0)
|
||||
|
||||
/* === Byte 1 (port 0): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack1_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port0, ack1_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack1_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port0, wait_ack1_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack1, port1_start)),
|
||||
|
||||
atom_label(ack1_received_port0)
|
||||
/* Read response ID byte — discarded for teaching scope (decode hardcoded). */
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* RX FIFO drain wait. Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel1_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port0, ack_released1_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel1_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port0, wait_ackrel1_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel1, port1_start)),
|
||||
|
||||
atom_label(ack_released1_port0)
|
||||
|
||||
/* === Byte 2 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack2_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port0, ack2_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack2_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port0, wait_ack2_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack2, port1_start)),
|
||||
|
||||
atom_label(ack2_received_port0)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel2_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port0, ack_released2_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel2_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port0, wait_ackrel2_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel2, port1_start)),
|
||||
|
||||
atom_label(ack_released2_port0)
|
||||
|
||||
/* === Byte 3 (port 0): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack3_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port0, ack3_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack3_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port0, wait_ack3_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ack3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ack3, port1_start)),
|
||||
|
||||
atom_label(ack3_received_port0)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel3_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port0, ack_released3_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel3_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port0, wait_ackrel3_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_ackrel3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_ackrel3, port1_start)),
|
||||
|
||||
atom_label(ack_released3_port0)
|
||||
|
||||
/* === Byte 4 (FINAL, port 0): send 0x00 + RX-not-empty wait + read final byte === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_rx4_port0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port0, rx4_received_port0)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_rx4_port0)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port0, wait_rx4_port0)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port0_from_rx4)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port0_from_rx4, port1_start)),
|
||||
|
||||
atom_label(rx4_received_port0)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
|
||||
|
||||
/* === RESPONSE DECODE (hardcoded for teaching scope) ===
|
||||
* Per the plan §"Phase 3 task 3.1" + spec §"Architecture":
|
||||
* - Full decode (buttons/axes from response bytes) is follow-up scope.
|
||||
* - Teaching scope: hardcode digital poll response.
|
||||
* status = PadSioStatus_Digital
|
||||
* buttons = 0x0000FFFF (no buttons pressed — placeholder)
|
||||
* axes = 0x80808080 (left_x=0x80, left_y=0x80, right_x=0x80, right_y=0x80)
|
||||
* attempt = 0
|
||||
*/
|
||||
atom_label(decode_port0)
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
|
||||
|
||||
/* /CS cleanup: raise /CS, clear stale status before exiting port 0. */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
|
||||
/* ============== PORT 1 SETUP ============== */
|
||||
/* Swap: R_PadCountdown holds sio_base_addr1; R_PadStatus holds state1. */
|
||||
atom_label(port1_start)
|
||||
add_u(R_PadSioBase, R_0, R_PadCountdown), /* sio_base_addr1 → R_PadSioBase */
|
||||
add_u(R_PadState, R_0, R_PadStatus), /* state1 → R_PadState */
|
||||
|
||||
/* ============== PORT 1 TRANSACTION (mirror of port 0) ============== */
|
||||
/* R_PadStatus + R_PadCountdown are no longer reserved (port 1 is the
|
||||
* last transaction); we still use R_T0/R_T1 as scratch to match port 0. */
|
||||
|
||||
/* 1. Cleanup: CTRL = 0x0010 (raise /CS, clear stale status) */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_BEFORE_TX = 1000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_BEFORE_TX),
|
||||
atom_label(settle_pre_port1)
|
||||
nop,
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_pre_port1, settle_pre_port1)),
|
||||
|
||||
/* 2. Port-select: CTRL = 0x0003 | (1 << 13) (port 1 select) */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
|
||||
or_i(R_T0, R_T0, 1 << 13), /* port 1 select bit (CTRL bit 13 = port select) */
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
/* Bounded by pad_SIO_SETTLE_AFTER_TX = 2000 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_SETTLE_AFTER_TX),
|
||||
atom_label(settle_post_port1)
|
||||
nop,
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(settle_post_port1, settle_post_port1)),
|
||||
|
||||
/* 3. Address byte (0x01) — send + RX-ready wait + read response + RX-drain confirmation */
|
||||
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack0_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack0_port1, ack0_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack0_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack0_port1, wait_ack0_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack0, end_atom)),
|
||||
|
||||
atom_label(ack0_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel0_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel0_port1, ack_released_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel0_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel0_port1, wait_ackrel0_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel0)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel0, end_atom)),
|
||||
|
||||
atom_label(ack_released_port1)
|
||||
|
||||
/* === Byte 1 (port 1): send 0x42 (cmd read) + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack1_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack1_port1, ack1_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack1_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack1_port1, wait_ack1_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack1, end_atom)),
|
||||
|
||||
atom_label(ack1_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel1_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel1_port1, ack_released1_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel1_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel1_port1, wait_ackrel1_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel1)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel1, end_atom)),
|
||||
|
||||
atom_label(ack_released1_port1)
|
||||
|
||||
/* === Byte 2 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack2_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack2_port1, ack2_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack2_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack2_port1, wait_ack2_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack2, end_atom)),
|
||||
|
||||
atom_label(ack2_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel2_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel2_port1, ack_released2_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel2_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel2_port1, wait_ackrel2_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel2)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel2, end_atom)),
|
||||
|
||||
atom_label(ack_released2_port1)
|
||||
|
||||
/* === Byte 3 (port 1): send 0x00 + RX-ready wait + read response + RX-drain confirmation === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ack3_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_ack3_port1, ack3_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ack3_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ack3_port1, wait_ack3_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ack3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ack3, end_atom)),
|
||||
|
||||
atom_label(ack3_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_ackrel3_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_equal(R_T0, R_0, atom_offset(wait_ackrel3_port1, ack_released3_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_ackrel3_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_ackrel3_port1, wait_ackrel3_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_ackrel3)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_ackrel3, end_atom)),
|
||||
|
||||
atom_label(ack_released3_port1)
|
||||
|
||||
/* === Byte 4 (FINAL, port 1): send 0x00 + RX-not-empty wait + read final byte === */
|
||||
/* Bounded by pad_SIO_WAIT_BUDGET = 4096 iterations. */
|
||||
add_ui(R_T0, R_0, 0x00),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(wait_rx4_port1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(wait_rx4_port1, rx4_received_port1)),
|
||||
add_ui_self(R_T1, -1),
|
||||
atom_label(continue_wait_rx4_port1)
|
||||
branch_ne(R_T1, R_0, atom_offset(continue_wait_rx4_port1, wait_rx4_port1)),
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Disconnected, R_PadState, R_T0),
|
||||
atom_label(skip_port1_from_rx4)
|
||||
branch_equal(R_0, R_0, atom_offset(skip_port1_from_rx4, end_atom)),
|
||||
|
||||
atom_label(rx4_received_port1)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET), /* discard final byte */
|
||||
|
||||
/* === RESPONSE DECODE (port 1) === */
|
||||
atom_label(decode_port1)
|
||||
mac_pad_sio_write_pad_state(PadSioStatus_Digital, R_PadState, R_T0),
|
||||
|
||||
/* /CS cleanup: raise /CS, clear stale status before exiting port 1. */
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
|
||||
atom_label(end_atom)
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_step wrap */
|
||||
|
||||
/* ----- pad_sio_diag_pin -----
|
||||
* Per-frame diagnostic counter. The caller binds R_DiagPinScratch to
|
||||
* scratch_for_atom_diag_pin for temporary gdb verification.
|
||||
*/
|
||||
#if 0 /* pad_sio_diag_pin — superseded (raw-SIO phase removed) */
|
||||
internal MipsAtom_(pad_sio_diag_pin) atom_info(atom_phase(pad_init)
|
||||
, atom_reads(R_T0, R_T1, R_DiagPinScratch)
|
||||
, atom_writes(R_T0, R_T1, R_DiagPinScratch)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
|
||||
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
|
||||
load_upper_i(R_DiagPinScratch, 0x8001),
|
||||
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
|
||||
|
||||
/* High half = 0xD1A6; low half increments once per atom invocation. */
|
||||
load_word(R_T1, R_DiagPinScratch, 0),
|
||||
nop,
|
||||
add_ui(R_T1, R_T1, 1),
|
||||
and_i(R_T0, R_T1, 0xFFFF),
|
||||
load_upper_i(R_T1, 0xD1A6),
|
||||
or_i(R_T1, R_T1, 0),
|
||||
or_u(R_T1, R_T1, R_T0),
|
||||
store_word(R_T1, R_DiagPinScratch, 0),
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_diag_pin wrap */
|
||||
|
||||
/* ----- pad_sio_diag_byte_exchange -----
|
||||
* Temporary two-byte wire probe: sends 0x01 and 0x42, then stores the
|
||||
* open-bus byte and response ID in scratch_for_atom_diag_pin.
|
||||
*/
|
||||
#if 0 /* pad_sio_diag_byte_exchange — superseded (raw-SIO phase removed) */
|
||||
internal MipsAtom_(pad_sio_diag_byte_exchange) atom_info(atom_phase(pad_init)
|
||||
, atom_reads(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
|
||||
, atom_writes(R_T0, R_T1, R_T2, R_PadSioBase, R_DiagPinScratch)
|
||||
) {
|
||||
/* FIX 2026-08-02: explicitly reload R_DiagPinScratch (R_T3 = $t3). Caller-saved
|
||||
* per O32 ABI; the rgcc binding in main() does not survive tape_run. */
|
||||
load_upper_i(R_DiagPinScratch, 0x8001),
|
||||
or_i(R_DiagPinScratch, R_DiagPinScratch, 0xC800),
|
||||
|
||||
/* FIX 2026-08-02: explicitly load KSEG1 base into R_PadSioBase (R_T6) at the
|
||||
* top. The rgcc() binding in main() does NOT survive the tape_run call
|
||||
* because R_T6 is caller-saved per the O32 ABI. */
|
||||
load_upper_i(R_PadSioBase, pad_IO_KSEG1_BASE >> 16),
|
||||
or_i(R_PadSioBase, R_PadSioBase, pad_IO_KSEG1_BASE & 0xFFFF),
|
||||
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_TX_ENABLE),
|
||||
or_i(R_T0, R_T0, pad_SIO_CTRL_DTR_CS),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
|
||||
add_ui(R_T0, R_0, pad_PROTO_ADDR),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(diag_wait_ack0)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack0, diag_ack0_done)),
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack0, diag_wait_ack0)),
|
||||
add_ui(R_T0, R_0, 0xDEADAC01),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack0, diag_timeout)),
|
||||
atom_label(diag_ack0_done)
|
||||
load_byte_u(R_T2, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T0, R_0, pad_PROTO_CMD_READ),
|
||||
store_byte(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
add_ui(R_T1, R_0, pad_SIO_WAIT_BUDGET),
|
||||
atom_label(diag_wait_ack1)
|
||||
load_half_u(R_T0, R_PadSioBase, pad_SIO_STAT_OFFSET),
|
||||
nop,
|
||||
and_i(R_T0, R_T0, pad_SIO_STAT_RX_NOT_EMPTY),
|
||||
branch_ne(R_T0, R_0, atom_offset(diag_wait_ack1, diag_ack1_done)),
|
||||
add_ui_self(R_T1, -1),
|
||||
branch_ne(R_T1, R_0, atom_offset(diag_wait_ack1, diag_wait_ack1)),
|
||||
add_ui(R_T0, R_0, 0xDEADAC02),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
branch_equal(R_0, R_0, atom_offset(diag_timeout_ack1, diag_timeout)),
|
||||
atom_label(diag_ack1_done)
|
||||
load_byte_u(R_T0, R_PadSioBase, pad_SIO_DATA_OFFSET),
|
||||
nop,
|
||||
shift_lleft(R_T0, R_T0, 8),
|
||||
or_u(R_T2, R_T2, R_T0),
|
||||
store_word(R_T2, R_DiagPinScratch, 0),
|
||||
atom_label(diag_success)
|
||||
branch_equal(R_0, R_0, atom_offset(diag_success, diag_done)),
|
||||
nop,
|
||||
atom_label(diag_timeout_ack0)
|
||||
add_ui(R_T0, R_0, 0xDEADAC01),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
atom_label(diag_timeout_ack1)
|
||||
add_ui(R_T0, R_0, 0xDEADAC02),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
atom_label(diag_timeout)
|
||||
add_ui(R_T0, R_0, 0xDEADACFF),
|
||||
store_word(R_T0, R_DiagPinScratch, 0),
|
||||
atom_label(diag_done)
|
||||
add_ui(R_T0, R_0, pad_SIO_CTRL_CLEANUP),
|
||||
store_half(R_T0, R_PadSioBase, pad_SIO_CTRL_OFFSET),
|
||||
mac_yield(),
|
||||
};
|
||||
#endif /* end pad_sio_diag_byte_exchange wrap */
|
||||
@@ -7,12 +7,12 @@ A rest from the usual.
|
||||
## Dependencies
|
||||
|
||||
I will be programming from a Windows 11 machine (may eventually try this on the Steam Deck...):
|
||||

|
||||
|
||||
[armips](https://github.com/Kingcom/armips)
|
||||
|
||||
* Supports doing bare-metal assembly for the ps1
|
||||
* `scoop install armips` or just clone and build..
|
||||
* Was used early in the course. Now I just use an macro asm dsl in C11.
|
||||
|
||||
[luajit-2.1](https://github.com/LuaJIT/LuaJIT.git)
|
||||
|
||||
@@ -73,3 +73,9 @@ scoop install luajit
|
||||

|
||||

|
||||

|
||||
|
||||
Win 11 machine:
|
||||
|
||||

|
||||
|
||||
Still haven't gotten around to trying this on linux...
|
||||
|
||||
+114
-21
@@ -180,29 +180,18 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
||||
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim.
|
||||
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math,
|
||||
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but
|
||||
# ZERO .o files pulled in — they were unused. The 5 kept libraries
|
||||
# (api, c, etc, gpu, gte) are required by the C-side calls in
|
||||
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||
$libraries = @(
|
||||
"api",
|
||||
"c",
|
||||
"c2",
|
||||
"card",
|
||||
"cd",
|
||||
"comb",
|
||||
"ds",
|
||||
"etc",
|
||||
"gpu",
|
||||
"gs",
|
||||
"gte",
|
||||
"gun",
|
||||
"hmd",
|
||||
"math",
|
||||
"mcrd",
|
||||
"mcx",
|
||||
"pad",
|
||||
"press",
|
||||
"sio",
|
||||
"snd",
|
||||
"spu",
|
||||
"tap"
|
||||
"gte"
|
||||
)
|
||||
foreach ($lib in $libraries) {
|
||||
$link_args += ($f_link_lib + $lib)
|
||||
@@ -350,10 +339,10 @@ function build-graphis_hello {
|
||||
}
|
||||
# build-graphis_hello
|
||||
|
||||
function build-gte_hello {
|
||||
function build-hello_gte {
|
||||
$includes += @()
|
||||
|
||||
$path_module = join-path $path_code 'gte_hello'
|
||||
$path_module = join-path $path_code 'hello_gte'
|
||||
$path_duffle = join-path $path_code 'duffle'
|
||||
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
@@ -458,8 +447,112 @@ function build-gte_hello {
|
||||
}
|
||||
}
|
||||
}
|
||||
build-gte_hello
|
||||
# build-hello_gte
|
||||
|
||||
function build-hello_joypad {
|
||||
$includes += @()
|
||||
|
||||
$path_module = join-path $path_code 'hello_joypad'
|
||||
$path_duffle = join-path $path_code 'duffle'
|
||||
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
|
||||
$src_c = join-path $path_module 'hello_joypad.c'
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||
|
||||
$assemble_args = @()
|
||||
$assemble_args += $f_debug
|
||||
$assemble_args += $f_optimize_none
|
||||
$assemble_args += ($f_include + $path_code)
|
||||
|
||||
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||
|
||||
$module_c = join-path $path_build 'hello_joypad_c.o'
|
||||
|
||||
$compile_args = @()
|
||||
$compile_args += $f_debug
|
||||
$compile_args += $f_optimize_none
|
||||
# $compile_args += $f_optimize_intrinsics
|
||||
# $compile_args += $f_optimize_size
|
||||
# $compile_args += $f_optimize_debug
|
||||
$compile_args += ($f_include + $path_code)
|
||||
compile-unit $src_c $module_c $includes $compile_args
|
||||
|
||||
$elf = join-path $path_build 'hello_joypad.elf'
|
||||
$exe = join-path $path_build 'hello_joypad.ps-exe'
|
||||
|
||||
$link_args = @()
|
||||
$link_args += $f_debug
|
||||
# $link_args += $f_optimize_size
|
||||
$link_modules = @(
|
||||
$module_asm_crt,
|
||||
$module_c
|
||||
)
|
||||
link-modules $link_modules $elf $link_args
|
||||
make-binary $elf $exe
|
||||
|
||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||
|
||||
$dwarfLineBin = join-path $path_build_gen 'hello_joypad.dwarf_line.bin'
|
||||
$dwarfArangesBin = join-path $path_build_gen 'hello_joypad.dwarf_aranges.bin'
|
||||
$dwarfRnglistsBin = join-path $path_build_gen 'hello_joypad.dwarf_rnglists.bin'
|
||||
$injectElf = join-path $path_build 'hello_joypad.dwarf-injected.elf'
|
||||
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
|
||||
{
|
||||
Write-Host "[build] DWARF-injecting $elf -> $injectElf"
|
||||
Copy-Item -LiteralPath $elf -Destination $injectElf -Force
|
||||
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
|
||||
$f_args = @(
|
||||
"--update-section=.debug_line=$dwarfLineBin",
|
||||
"--update-section=.debug_aranges=$dwarfArangesBin",
|
||||
"--update-section=.debug_rnglists=$dwarfRnglistsBin"
|
||||
)
|
||||
& $Objcopy @f_args $injectElf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] objcopy F' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
||||
return;
|
||||
}
|
||||
|
||||
$dwarfInfoBin = join-path $path_build_gen 'hello_joypad.dwarf_info.bin'
|
||||
$dwarfAbbrevBin = join-path $path_build_gen 'hello_joypad.dwarf_abbrev.bin'
|
||||
$dwarfStrBin = join-path $path_build_gen 'hello_joypad.dwarf_str.bin'
|
||||
$dwarfLocBin = join-path $path_build_gen 'hello_joypad.dwarf_loc.bin'
|
||||
$dwarfLoclistsBin = join-path $path_build_gen 'hello_joypad.dwarf_loclists.bin'
|
||||
$g_args = @(
|
||||
"--update-section=.debug_info=$dwarfInfoBin",
|
||||
"--update-section=.debug_abbrev=$dwarfAbbrevBin",
|
||||
"--update-section=.debug_str=$dwarfStrBin",
|
||||
"--add-section=.debug_loc=$dwarfLocBin",
|
||||
"--add-section=.debug_loclists=$dwarfLoclistsBin"
|
||||
)
|
||||
& $Objcopy @g_args $injectElf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] objcopy G' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
||||
return;
|
||||
}
|
||||
|
||||
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
||||
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
||||
# The original ELF and PS-EXE remain byte/flag unchanged.
|
||||
& $Objcopy `
|
||||
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
||||
--set-section-flags ".data=alloc,load,data,code,contents" `
|
||||
$injectElf 2>&1 | Out-Null
|
||||
if ($LASTEXITCODE -ne 0) {
|
||||
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
|
||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
||||
}
|
||||
else {
|
||||
Write-Host "[build] DWARF-injected ELF: $injectElf"
|
||||
}
|
||||
}
|
||||
}
|
||||
build-hello_joypad
|
||||
|
||||
# NO idea if this works yet...
|
||||
function Send-ToEmulator { param( [string]$exePath )
|
||||
|
||||
+54
-67
@@ -920,18 +920,26 @@ function M.tokenize_body(body)
|
||||
while scan <= len do
|
||||
local c = body:byte(scan)
|
||||
-- Terminator bytes (delimit a token at the top level): ',' = 0x2C, '\n' = 0x0A, ';' = 0x3B.
|
||||
-- These also appear as separators between argument lists inside the parens/braces/brackets,
|
||||
-- These also appear as separators between argument lists inside the parens/braces/brackets,
|
||||
-- so we stop the scan when we hit any of them.
|
||||
if c == BYTE_COMMA then break end
|
||||
if c == BYTE_NEWLINE then break end
|
||||
if c == BYTE_SEMI then break end
|
||||
-- Line-comment '// ... \n' (0x2F 0x2F): skip to (and past) the next newline, or to end-of-body.
|
||||
if c == BYTE_SLASH and body:byte(scan + 1) == BYTE_SLASH then
|
||||
local nl = M.find_byte(body, BYTE_NEWLINE, scan)
|
||||
scan = nl and (nl + 1) or (len + 1)
|
||||
-- Block-comment '/* ... */' (0x2F 0x2A): skip to (and past) the matching '*/', or to end-of-body.
|
||||
elseif c == BYTE_SLASH and body:byte(scan + 1) == BYTE_STAR then
|
||||
local close = body:find("*/", scan + 2, true)
|
||||
scan = close and (close + 2) or (len + 1)
|
||||
-- Group opener bytes (consume the balanced group via the matching reader): '(' = 0x28, '{' = 0x7B, '[' = 0x5B.
|
||||
if c == BYTE_OPEN_PAREN then local _, a = M.read_parens (body, scan); scan = a
|
||||
elseif c == BYTE_OPEN_PAREN then local _, a = M.read_parens (body, scan); scan = a
|
||||
elseif c == BYTE_OPEN_BRACE then local _, a = M.read_braces (body, scan); scan = a
|
||||
elseif c == BYTE_OPEN_BRACK then local _, a = M.read_brackets (body, scan); scan = a
|
||||
-- String-literal byte ('"' = 0x22 or '\'' = 0x27): skip past the quoted region in one shot.
|
||||
elseif c == BYTE_DQUOTE or c == BYTE_SQUOTE then
|
||||
scan = M.skip_str_or_cmt(body, scan) + 1
|
||||
scan = (M.skip_str_or_cmt(body, scan) or scan) + 1
|
||||
else
|
||||
scan = scan + 1
|
||||
end
|
||||
@@ -1180,10 +1188,9 @@ M.GTE_COMMAND_INPUTS = {
|
||||
-- * "mac_result" : generic MAC output (nclip, op, mvmva)
|
||||
--
|
||||
-- Consumers:
|
||||
-- * passes/static_analysis.lua::analyze_hardware_relations (the walker reads this after a GTE command to update
|
||||
-- `forward_state.post_command_roles` for `gte_result_position`).
|
||||
-- * passes/static_analysis.lua::check_gte_result_position (per-atom CHECK_RULES reader; renders role mismatches).
|
||||
-- This table is consumed by the hardware-relation analyzer and result-position check.
|
||||
-- * passes/static_analysis.lua::analyze_hardware_relations (the walker reads this after a GTE command to update `forward_state.post_command_roles` for `gte_role_mismatch`).
|
||||
-- * passes/static_analysis.lua::check_gte_role_mismatch (per-atom CHECK_RULES reader; renders role mismatches).
|
||||
-- This table is consumed by the hardware-relation analyzer and the gte_role_mismatch check.
|
||||
M.GTE_COMMAND_OUTPUTS = {
|
||||
-- RTPS: writes one screen coordinate (the perspective-divide result) into C2_SXY2.
|
||||
-- The FIFO side effects leave SXY0 / SXY1 untouched, so `latest_screen_xy` is C2_SXY2.
|
||||
@@ -1293,32 +1300,8 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
|
||||
},
|
||||
}
|
||||
|
||||
-- GTE component result contracts (immutable; keyed by bare component name).
|
||||
--
|
||||
-- Register-role claims that the `_post_<cmd>` suffix alone cannot infer live here.
|
||||
-- The bare name (the component name stripped of the `_post_<cmd>` suffix) is the key; the row carries the expected
|
||||
-- command, the expected role, and the expected C2 register.
|
||||
--
|
||||
-- Known rows:
|
||||
-- * `gte_store_g4_p3_post_rtps`: post-RTPS polygon-emit slot reads the newest projected screen coordinate from C2_SXY2.
|
||||
-- C2_SXY0 is wrong (C2_SXY0 is an older FIFO entry, never the newest post-RTPS result).
|
||||
--
|
||||
-- Unknown `_post_<cmd>` components (a `<name>_post_<cmd>`-suffixed component whose bare `<name>` is not a row key) emit one
|
||||
-- `table_gap` info finding so downstream consumers can detect when the contract table is incomplete for an authored atom body.
|
||||
--
|
||||
-- Consumers:
|
||||
-- * passes/static_analysis.lua::check_gte_result_position (renders result-position findings).
|
||||
-- * passes/static_analysis.lua::emit_table_gap_warning (called once per atom body; surfaces the missing-row diagnostic).
|
||||
-- This table is consumed by the result-position check.
|
||||
M.GTE_COMPONENT_RESULT_CONTRACTS = {
|
||||
-- Post-RTPS g4 p3 store contract: writes the latest screen XY (C2_SXY2) into the primitive's p3 slot.
|
||||
-- Reading from C2_SXY0 is a semantic mismatch — C2_SXY0 is the oldest post-RTPS SXY, not the newest one.
|
||||
["gte_store_g4_p3_post_rtps"] = {
|
||||
command = "gte_cmdw_rtps",
|
||||
role = "latest_screen_xy",
|
||||
register = "C2_SXY2",
|
||||
},
|
||||
}
|
||||
-- GTE component result contracts were removed: the `_post_<cmd>` naming convention was a soft convention
|
||||
-- (the user did not want it formalized via static-analysis enforcement). A proper `atom_info` directive for ordering semantics is a future TODO.
|
||||
|
||||
-- Operand-class table for the COP2->GPR load-delay check.
|
||||
--
|
||||
@@ -1341,13 +1324,13 @@ M.OPERAND_READ_POSITIONS = {
|
||||
["sub_s"] = {1, 2, 3},
|
||||
["sub_u"] = {1, 2, 3},
|
||||
["and_i"] = {1, 2},
|
||||
["and_u"] = {1, 2, 3},
|
||||
["and"] = {1, 2, 3},
|
||||
["or_i"] = {1, 2},
|
||||
["or_i_self"] = {1},
|
||||
["or_u"] = {1, 2, 3},
|
||||
["or_u_self"] = {1, 2},
|
||||
["or"] = {1, 2, 3},
|
||||
["or_self"] = {1, 2},
|
||||
["xor_i"] = {1, 2},
|
||||
["xor_u"] = {1, 2, 3},
|
||||
["xor"] = {1, 2, 3},
|
||||
["slt_s"] = {1, 2, 3},
|
||||
["slt_u"] = {1, 2, 3},
|
||||
["slt_si"] = {1, 2},
|
||||
@@ -1447,20 +1430,21 @@ M.GP0_CMD_BY_SHAPE = {
|
||||
["g4"] = 0x38, ["gt4"] = 0x3C,
|
||||
}
|
||||
|
||||
-- TODO(Ed): REMOVE THIS HARDCODE, THIS SHOULD BE RESOLVED AUTOMATICALLY
|
||||
-- Per-macro prim-buffer contribution: how many 32-bit words each macro writes to the primitive being built in main RAM.
|
||||
-- (This counts RAM-side prim-buffer words, not .text instruction words.)
|
||||
-- The sum across `mac_format_X_color` + `mac_gte_store_X_post_*` + `mac_insert_ot_tag_X` calls in an atom body must equal
|
||||
-- `GP0_CMD_SIZE[GP0_CMD_BY_SHAPE[shape]]`.
|
||||
M.GP0_MACRO_CONTRIB = {
|
||||
["mac_format_f3_color"] = 1,
|
||||
["mac_format_g3_color"] = 3,
|
||||
["mac_format_g4_color"] = 4,
|
||||
["mac_gte_store_f3_post_rtpt"] = 3,
|
||||
["mac_gte_store_g3_post_rtpt"] = 3,
|
||||
["mac_gte_store_g4_p012_post_rtpt_pre_rtps"] = 3,
|
||||
["mac_gte_store_g4_p3_post_rtps"] = 1,
|
||||
["mac_insert_ot_tag_f3"] = 1,
|
||||
["mac_insert_ot_tag_g4"] = 1,
|
||||
["mac_format_f3_color"] = 1,
|
||||
["mac_format_g3_color"] = 3,
|
||||
["mac_format_g4_color"] = 4,
|
||||
["mac_gte_store_f3"] = 3,
|
||||
["mac_gte_store_g3"] = 3,
|
||||
["mac_gte_store_g4_p012"] = 3,
|
||||
["mac_gte_store_g4_p3"] = 1,
|
||||
["mac_insert_ot_tag_f3"] = 1,
|
||||
["mac_insert_ot_tag_g4"] = 1,
|
||||
}
|
||||
|
||||
-- Per-macro cycle cost (best-case, no stalls). Used by the static-analysis pass to emit per-atom cycle budgets.
|
||||
@@ -1493,14 +1477,14 @@ M.INSTRUCTION_LATENCY = {
|
||||
-- CPU ALU (single-cycle R3000A ops)
|
||||
["nop"] = 1,
|
||||
["nop2"] = 2,
|
||||
["add_ui"] = 1, ["add_ui_self"] = 1,
|
||||
["add_s"] = 1, ["add_si"] = 1,
|
||||
["add_u"] = 1, ["add_u_self"] = 1,
|
||||
["sub_u"] = 1, ["sub_s"] = 1,
|
||||
["and_i"] = 1, ["and_u"] = 1,
|
||||
["or_i"] = 1, ["or_i_self"] = 1,
|
||||
["or_u"] = 1, ["or_u_self"] = 1,
|
||||
["xor_i"] = 1, ["xor_u"] = 1,
|
||||
["add_ui"] = 1, ["add_ui_self"] = 1,
|
||||
["add_s"] = 1, ["add_si"] = 1,
|
||||
["add_u"] = 1, ["add_u_self"] = 1,
|
||||
["sub_u"] = 1, ["sub_s"] = 1,
|
||||
["and_i"] = 1, ["and"] = 1,
|
||||
["or_i"] = 1, ["or_i_self"] = 1,
|
||||
["or_u"] = 1, ["or_u_self"] = 1,
|
||||
["xor_i"] = 1, ["xor_u"] = 1,
|
||||
["nor_u"] = 1,
|
||||
["shift_lleft"] = 1, ["shift_lleft_self"] = 1,
|
||||
["shift_lright"] = 1,
|
||||
@@ -1583,20 +1567,23 @@ M.INSTRUCTION_LATENCY = {
|
||||
["gte_load_v1"] = 2,
|
||||
["gte_load_v2"] = 2,
|
||||
["gte_load_v0v1v2"] = 6,
|
||||
|
||||
-- TODO(Ed): REMOVE THIS HARDCODE, THIS SHOULD BE RESOLVED AUTOMATICALLY
|
||||
-- mac_* helpers (cycle cost = sum of the expanded instructions)
|
||||
-- mac_yield transfers control; cycle budget is 0 (the next atom absorbs the cost).
|
||||
["mac_yield"] = 0,
|
||||
["mac_pack_color_word"] = 3, -- lui + ori + sw
|
||||
["mac_format_f3_color"] = 3, -- = mac_pack_color_word
|
||||
["mac_format_g4_color"] = 12, -- 4 x mac_pack_color_word
|
||||
["mac_load_tri_indices"] = 3, -- 3 x lhu
|
||||
["mac_gte_load_tri_verts"] = 18, -- 3 x {sll, addu, lw, lw, mtc2, mtc2}
|
||||
["mac_gte_store_f3_post_rtpt"] = 3,
|
||||
["mac_gte_store_g3_post_rtpt"] = 3,
|
||||
["mac_gte_store_g4_p012_post_rtpt_pre_rtps"] = 3,
|
||||
["mac_gte_store_g4_p3_post_rtps"] = 1,
|
||||
["mac_insert_ot_tag_f3"] = 11, -- 11 .word slots in the macro body
|
||||
["mac_insert_ot_tag_g4"] = 11,
|
||||
["mac_yield"] = 0,
|
||||
["mac_pack_color_word"] = 3, -- lui + ori + sw
|
||||
["mac_format_f3_color"] = 3, -- = mac_pack_color_word
|
||||
["mac_format_g4_color"] = 12, -- 4 x mac_pack_color_word
|
||||
["mac_load_tri_indices"] = 3, -- 3 x lhu
|
||||
["mac_gte_load_tri_verts"] = 18, -- 3 x {sll, addu, lw, lw, mtc2, mtc2}
|
||||
["mac_gte_store_f3"] = 3,
|
||||
["mac_gte_store_g3"] = 3,
|
||||
["mac_gte_store_g4_p012"] = 3,
|
||||
["mac_gte_store_g4_p3"] = 1,
|
||||
["mac_insert_ot_tag_f3"] = 11, -- 11 .word slots in the macro body
|
||||
["mac_insert_ot_tag_g4"] = 11,
|
||||
|
||||
-- Annotation markers (emit no code; pure metaprogram hints)
|
||||
["atom_label"] = 0,
|
||||
["atom_offset"] = 0,
|
||||
@@ -1973,7 +1960,7 @@ M.GPR_VALUE_RULES = {
|
||||
-- Present register-form self variants. They are included here so a
|
||||
-- known value is not needlessly lost when these encoders are used.
|
||||
add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, },
|
||||
or_u_self = { op = "or_u", dest = 1, sources = {1, 2}, },
|
||||
or_u_self = { op = "or", dest = 1, sources = {1, 2}, },
|
||||
shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, },
|
||||
}
|
||||
|
||||
|
||||
+262
-1
@@ -193,7 +193,7 @@ M.DWARF_LINE_OPS = {
|
||||
DW_LNE_set_address = 2, -- spec: §6.2.5.3
|
||||
-- Standard opcode header (§6.2.5.1)
|
||||
-- opcode_base + line_range are 1-byte header fields; hex so they map
|
||||
-- directly to their position in the line-program header byte sequence.
|
||||
-- directly to the line-program header byte sequence.
|
||||
-- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics.
|
||||
opcode_base = 0x0D,
|
||||
line_base = -5,
|
||||
@@ -204,6 +204,44 @@ M.DWARF_LINE_OPS = {
|
||||
set_address_payload_size = 0x05, -- size = sub_opcode(1) + addr(4)
|
||||
}
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- DWARF5 .debug_line (per DWARF5 spec §6.2.4 — Line Number Program Header)
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- All offsets are zero-based wire offsets from the start of the unit body
|
||||
-- (i.e. AFTER unit_length has been read and unit_length bytes skipped past unit_length's 4 bytes).
|
||||
--
|
||||
-- The DWARF3/4 line-program format differs:
|
||||
-- - It omits `address_size` (DWARF3 §6.2.4) + `segment_selector_size` (DWARF5 §6.2.4).
|
||||
-- - It uses null-terminated string lists for `include_directories` + `file_names`
|
||||
-- (vs. DWARF5's format_count + fields-list shape).
|
||||
-- These are documented inline at each parse site in read_line_unit_file_table below.
|
||||
|
||||
--- spec: DWARF5 spec §6.2.4 (Line Number Program Header — version >= 5)
|
||||
M.DWARF5_DEBUG_LINE = {
|
||||
-- Header fields (zero-based, AFTER unit_length has been read).
|
||||
version_offset_post_il = 0x00, -- 2-byte LE; expected = 5
|
||||
addr_size_offset = 0x02, -- 1 byte; expected = 4
|
||||
seg_size_offset = 0x03, -- 1 byte; expected = 0
|
||||
header_length_offset = 0x04, -- 4-byte LE; length of program-header content that follows
|
||||
program_header_start = 0x08, -- first byte of program-header content (after the 8 fixed bytes)
|
||||
|
||||
-- Per-form byte widths (used when reading directory / file-name entries).
|
||||
form_addr_bytes = 0x04, -- DW_FORM_addr (32-bit) | DW_FORM_data4
|
||||
form_strp_bytes = 0x04, -- DW_FORM_line_strp / DW_FORM_strp / DW_FORM_strp_sup
|
||||
form_data16_bytes = 0x10, -- DW_FORM_data16 (MD5)
|
||||
|
||||
-- DWARF5 form codes (subset used in line-program directory + file tables).
|
||||
form_line_strp = 0x1A, -- DWARF5 §7.5.6 — DW_FORM_line_strp (4-byte offset into .debug_line_str)
|
||||
form_string = 0x08, -- DWARF4-compatible fallback (inline null-terminated; not in .debug_line_str)
|
||||
form_udata = 0x0F, -- DW_FORM_udata (ULEB)
|
||||
form_data16 = 0x18, -- DW_FORM_data16 (16-byte MD5; gcc emits this for split debug info)
|
||||
|
||||
-- DWARF5 content-tag codes (DW_LNCT_* from §6.2.4.1 + §6.2.4.2).
|
||||
lnct_path = 0x01,
|
||||
lnct_directory_index = 0x02,
|
||||
lnct_md5 = 0x05, -- gcc with MD5 in file name table (rare)
|
||||
}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- I/O helpers: little-endian byte read/write
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -787,6 +825,229 @@ function M.sleb128_size(n)
|
||||
return bytes
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- DWARF5 line-program file-table reader
|
||||
-- ════════════════════════════════════════════
|
||||
|
||||
--- Read every line-program unit in `.debug_line` and produce one entry per file across all units.
|
||||
--- Returns three parallel maps keyed by 1-based file index.
|
||||
---
|
||||
--- Wire format notes:
|
||||
--- * The `.debug_line` section may contain MULTIPLE line-program units
|
||||
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
|
||||
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
|
||||
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom
|
||||
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
|
||||
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
|
||||
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
|
||||
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index`
|
||||
--- emits 2 forms: path + dir_index). The helper supports:
|
||||
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
|
||||
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
|
||||
--- - DW_FORM_udata (ULEB128)
|
||||
--- - DW_FORM_data16 (16-byte MD5; ignored — skip the form's bytes)
|
||||
--- * Symlink-canonicalisation: each path's `paths[i]` is stored verbatim from the wire
|
||||
--- (mixed `/` and `\` accepted; the basename is taken via the last path separator). Caller normalises as needed.
|
||||
---
|
||||
--- Behavior on failure: writes to stderr and returns nil.
|
||||
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
|
||||
--- downstream `resolve_provenance_file_index(path)` consumers
|
||||
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
|
||||
--- consult the map directly.
|
||||
---
|
||||
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice;
|
||||
--- both shapes work since the splice preserves `.debug_line`)
|
||||
--- @return table|nil, table|nil, table|nil
|
||||
--- basename_to_index: { [basename] = 1-based-per-unit-file-index, ... }
|
||||
--- basenames: { [1-based-per-unit-file-index] = basename, ... }
|
||||
--- paths: { [1-based-per-unit-file-index] = full path (mixed slashes), ... }
|
||||
function M.read_line_unit_file_table(elf_path)
|
||||
local sections = M.read_elf_sections(elf_path, { ".debug_line", ".debug_line_str" })
|
||||
local line = sections[".debug_line"]
|
||||
local lstr = sections[".debug_line_str"] or ""
|
||||
if not line or line == "" then
|
||||
io.stderr:write("[elf_dwarf.read_line_unit_file_table] no .debug_line section in: " .. tostring(elf_path) .. "\n")
|
||||
return nil
|
||||
end
|
||||
|
||||
local basenames = {}
|
||||
local basename_to_index = {}
|
||||
local paths = {}
|
||||
|
||||
--- Read one form-code's bytes from `buf` at position `p` according to `form`.
|
||||
--- Returns (value, after) where `value` is:
|
||||
--- * the resolved string (DW_FORM_line_strp / DW_FORM_string)
|
||||
--- * the ULEB128 number (DW_FORM_udata)
|
||||
--- * nil + skip-bytes (DW_FORM_data16; we don't surface the MD5)
|
||||
local function read_form(buf, lstr_buf, p, form)
|
||||
if form == M.DWARF5_DEBUG_LINE.form_line_strp then
|
||||
local strp = M.read_u32_le(buf, p)
|
||||
local end_pos = lstr_buf:find("\0", strp + 1, true) or (#lstr_buf + 1)
|
||||
return lstr_buf:sub(strp + 1, end_pos - 1), p + M.DWARF5_DEBUG_LINE.form_strp_bytes
|
||||
elseif form == M.DWARF5_DEBUG_LINE.form_string then
|
||||
local nul = buf:find("\0", p + 1, true) or (#buf + 1)
|
||||
return buf:sub(p + 1, nul - 1), nul
|
||||
elseif form == M.DWARF5_DEBUG_LINE.form_udata then
|
||||
local v, after = M.read_uleb128_at(buf, p)
|
||||
return v, after
|
||||
elseif form == M.DWARF5_DEBUG_LINE.form_data16 then
|
||||
return nil, p + M.DWARF5_DEBUG_LINE.form_data16_bytes
|
||||
else
|
||||
-- Unsupported form in a directory/file-table entry: best-effort skip.
|
||||
-- We do NOT stderr-write because the crt0.s DWARF5 line unit (gcc-as emitted) uses DW_FORM_addr (0x01) for what is effectively a path entry,
|
||||
-- which is non-standard.
|
||||
-- The C-unit's DWARF3 paths are read via the parallel DWARF3 path and never see this error.
|
||||
-- Callers should consult `basename_to_index` for the paths they care about and ignore this unit if it produced none.
|
||||
return nil, p
|
||||
end
|
||||
end
|
||||
|
||||
--- Parse one DWARF-version-3-style unit (DWARF3/4 line program; gcc default in the PS1 toolchain still emits DWARF3 for line programs in `-g` mode).
|
||||
--- Layout: null-terminated directory list, then path(null) + dir_idx(ULEB) + time(ULEB) + size(ULEB) file entries terminated by an empty null.
|
||||
--- `content_start` = zero-based wire offset of the first byte of program-header content (after version + header_length fields).
|
||||
--- @return unit_basenames { [idx_in_unit_1_based] = basename }
|
||||
--- @return unit_paths { [idx_in_unit_1_based] = full path }
|
||||
local function parse_dwarf3_unit(buf, content_start, body_end)
|
||||
local up = content_start
|
||||
-- 5 fixed bytes: min_insn, default_is, line_base (signed), line_range, opcode_base
|
||||
up = up + 5
|
||||
local opcode_base = buf:byte(content_start + 5)
|
||||
up = up + (opcode_base - 1) -- std_opcode_lengths
|
||||
local dirs = {}
|
||||
while up < body_end do
|
||||
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
|
||||
if nul > body_end then break end
|
||||
local len = nul - up - 1
|
||||
if len == 0 then up = nul break end
|
||||
dirs[#dirs + 1] = buf:sub(up + 1, nul - 1)
|
||||
up = nul
|
||||
end
|
||||
local unit_basenames = {}
|
||||
local unit_paths = {}
|
||||
while up < body_end do
|
||||
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
|
||||
if nul > body_end or nul == up + 1 then up = nul break end
|
||||
local path = buf:sub(up + 1, nul - 1)
|
||||
up = nul
|
||||
local didx, up_next = M.read_uleb128_at(buf, up); up = up_next
|
||||
local _time, up_next2 = M.read_uleb128_at(buf, up); up = up_next2
|
||||
local _size, up_next3 = M.read_uleb128_at(buf, up); up = up_next3
|
||||
local idx = #unit_basenames + 1
|
||||
local bs = path:match("[^/\\]+$") or path
|
||||
unit_paths[idx] = path
|
||||
unit_basenames[idx] = bs
|
||||
dirs[1] = dirs[1] or "" -- safety: gcc emits "" sentinel dir at 0
|
||||
if didx > 0 and dirs[didx] then
|
||||
unit_paths[idx] = dirs[didx] .. "/" .. path
|
||||
end
|
||||
end
|
||||
return unit_basenames, unit_paths
|
||||
end
|
||||
|
||||
--- Parse one DWARF-version-5-style unit (DWARF5 line program; used by modern gcc with `-gdwarf-5`).
|
||||
--- `content_start` is the first byte of program-header content (after the 8 fixed bytes version+addr_size+seg_size+header_length).
|
||||
--- @return same shape as parse_dwarf3_unit
|
||||
local function parse_dwarf5_unit(buf, lstr_buf, content_start, body_end)
|
||||
local up = content_start
|
||||
-- 6 fixed bytes: min_insn, max_ops_per_insn, default_is, line_base, line_range, opcode_base
|
||||
up = up + 6
|
||||
local opcode_base = buf:byte(content_start + 6)
|
||||
up = up + (opcode_base - 1) -- std_opcode_lengths
|
||||
-- directories
|
||||
local dir_format_count, after = M.read_uleb128_at(buf, up); up = after
|
||||
local dir_formats = {}
|
||||
for i = 1, dir_format_count do
|
||||
local f, a2 = M.read_uleb128_at(buf, up); up = a2
|
||||
dir_formats[i] = f
|
||||
end
|
||||
local dir_count, a3 = M.read_uleb128_at(buf, up); up = a3
|
||||
local dirs = {}
|
||||
for i = 1, dir_count do
|
||||
local combined = ""
|
||||
for j = 1, dir_format_count do
|
||||
local v, a4 = read_form(buf, lstr_buf, up, dir_formats[j])
|
||||
up = a4
|
||||
if j == 1 and type(v) == "string" then combined = v end
|
||||
end
|
||||
dirs[i] = combined
|
||||
end
|
||||
-- file names
|
||||
local file_format_count, after2 = M.read_uleb128_at(buf, up); up = after2
|
||||
local file_formats = {}
|
||||
for i = 1, file_format_count do
|
||||
local f, a2 = M.read_uleb128_at(buf, up); up = a2
|
||||
file_formats[i] = f
|
||||
end
|
||||
local file_count, a3 = M.read_uleb128_at(buf, up); up = a3
|
||||
local unit_basenames = {}
|
||||
local unit_paths = {}
|
||||
for i = 1, file_count do
|
||||
local combined = ""
|
||||
local didx = 0
|
||||
for j = 1, file_format_count do
|
||||
local v, a4 = read_form(buf, lstr_buf, up, file_formats[j])
|
||||
up = a4
|
||||
if j == 1 and type(v) == "string" then combined = v end
|
||||
if j == 2 and type(v) == "number" then didx = v end
|
||||
end
|
||||
local idx = #unit_basenames + 1
|
||||
local bs = combined:match("[^/\\]+$") or combined
|
||||
unit_paths[idx] = combined
|
||||
unit_basenames[idx] = bs
|
||||
if didx > 0 and dirs[didx] then
|
||||
unit_paths[idx] = dirs[didx] .. "/" .. combined
|
||||
end
|
||||
end
|
||||
return unit_basenames, unit_paths
|
||||
end
|
||||
|
||||
--- Walk every line-program unit in the section.
|
||||
local p = 0
|
||||
local section_end = #line
|
||||
while p + 4 <= section_end do
|
||||
local unit_length = M.read_u32_le(line, p)
|
||||
if unit_length == 0xFFFFFFFF then
|
||||
io.stderr:write("[elf_dwarf.read_line_unit_file_table] 64-bit DWARF (initial-length 0xFFFFFFFF); not supported\n")
|
||||
return nil
|
||||
end
|
||||
local body_start = p + 4
|
||||
local body_end = p + 4 + unit_length
|
||||
if body_end > section_end then break end
|
||||
local version = M.read_u16_le(line, body_start)
|
||||
local unit_basenames, unit_paths
|
||||
if version >= 5 then
|
||||
-- DWARF5 header: version(2) + addr_size(1) + seg_size(1) + header_length(4) + content
|
||||
local header_length_offset = body_start + 6 -- past version(2) + addr_size(1) + seg_size(1) - wait that's wrong; past hdr len is at +6
|
||||
local content_start = body_start + 8 -- past version(2) + addr_size(1) + seg_size(1) + header_length(4)
|
||||
unit_basenames, unit_paths = parse_dwarf5_unit(line, lstr, content_start, body_end)
|
||||
elseif version >= 2 then
|
||||
-- DWARF2/3/4 header: version(2) + header_length(4) + content
|
||||
local content_start = body_start + 6 -- past version(2) + header_length(4)
|
||||
unit_basenames, unit_paths = parse_dwarf3_unit(line, content_start, body_end)
|
||||
else
|
||||
io.stderr:write(string.format("[elf_dwarf.read_line_unit_file_table] unsupported DWARF version %d (offset 0x%x)\n", version, p))
|
||||
p = body_end
|
||||
goto continue
|
||||
end
|
||||
-- Per-unit 1-based file indices are aligned with `inv.call_file` values because the metaprogram emits `DW_LNS_set_file` with the per-unit index.
|
||||
-- When multiple units are present (crt0.s + C unit), the per-unit index in each unit matches the metaprogram's intent (gcc always sets file in unit-local terms).
|
||||
-- We therefore store directly without global re-indexing; the caller is responsible for knowing which unit the file-index applies to.
|
||||
-- For DWARF3 (C unit is the unit that matters for atom line tables), this matches.
|
||||
-- For DWARF5 (crt0.s + C unit), each carries its own per-unit file-table map;
|
||||
-- the atom-side DW_LNS_set_file(N) refers to the C unit's indices, NOT crt0.s's.
|
||||
-- Since the C unit is the one with full include_directories + 12 entries, we can use it directly.
|
||||
for idx, bs in pairs(unit_basenames) do
|
||||
basenames[idx] = bs
|
||||
paths[idx] = unit_paths[idx]
|
||||
basename_to_index[bs] = idx
|
||||
end
|
||||
p = body_end
|
||||
::continue::
|
||||
end
|
||||
|
||||
return basename_to_index, basenames, paths
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- I/O helpers: atoms source-map + native directory glob
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
@@ -3,16 +3,16 @@
|
||||
# Wrapper for the tape-atom step-debug helpers.
|
||||
# The 9 user commands are defined here as STUBS (degraded-state messages).
|
||||
# The real implementations + the per-atom data tables are emitted by `passes/atoms_source_map.lua`
|
||||
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gen/gdb_tape_atoms_runtime.gdb`.
|
||||
# (post-link invocation: `ps1_meta.lua --atoms-source-map --gdb-runtime --elf <elf>`) into `build/gdb_tape_atoms_runtime.gdb`.
|
||||
# Sourcing that file RE-DEFINES the commands with real implementations.
|
||||
#
|
||||
# If `build/gen/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
|
||||
# If `build/gdb_tape_atoms_runtime.gdb` is missing or stale, the stubs remain (E1: no source map).
|
||||
# The user just needs to re-run `build_psyq.ps1` to regenerate.
|
||||
|
||||
# ── Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ──
|
||||
# ?? Stub commands (defined here so they're always present, even if the runtime file is missing). The runtime file overrides these if sourced. ??
|
||||
|
||||
define tape_atoms
|
||||
echo "[gdb_tape_atoms] STUB: runtime file build/gen/gdb_tape_atoms_runtime.gdb not found."
|
||||
echo "[gdb_tape_atoms] STUB: runtime file build/gdb_tape_atoms_runtime.gdb not found."
|
||||
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||
end
|
||||
document tape_atoms
|
||||
@@ -21,35 +21,35 @@ document tape_atoms
|
||||
end
|
||||
|
||||
define break_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document break_atom
|
||||
Set a breakpoint at the start of tape atom <name>. STUB state.
|
||||
end
|
||||
|
||||
define step_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document step_atom
|
||||
Resume execution until the next atom boundary. STUB state.
|
||||
end
|
||||
|
||||
define next_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document next_atom
|
||||
Alias for step_atom. STUB state.
|
||||
end
|
||||
|
||||
define where_in_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document where_in_atom
|
||||
Report current atom name, .rodata addr, word offset, and source line (if known). STUB state.
|
||||
end
|
||||
|
||||
define stepi_inside_atom
|
||||
echo "[gdb_tape_atoms] STUB: build/gen/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
echo "[gdb_tape_atoms] STUB: build/gdb_tape_atoms_runtime.gdb not sourced. Run build_psyq.ps1."
|
||||
end
|
||||
document stepi_inside_atom
|
||||
One MIPS-instruction step, then where_in_atom. STUB state.
|
||||
@@ -89,17 +89,17 @@ document wave_ctx
|
||||
end
|
||||
|
||||
|
||||
# ── Source the runtime file (re-defines commands with real impls + data). ──
|
||||
# ?? Source the runtime file (re-defines commands with real impls + data). ??
|
||||
|
||||
# Try to source from project-root-relative path first (the typical case).
|
||||
# If the user is in a different CWD, the source will fail and stubs remain.
|
||||
# The runtime file path is computed relative to the ELF's source map convention (build/gen/gdb_tape_atoms_runtime.gdb).
|
||||
# The runtime file path is computed relative to the ELF's source map convention (build/gdb_tape_atoms_runtime.gdb).
|
||||
echo [gdb_tape_atoms] Wrapper loaded. Sourcing runtime file...
|
||||
# Suppress the "Redefine command" prompts that would otherwise appear when the runtime file overrides the 9 stub commands defined above.
|
||||
# The runtime's `define` blocks are intended to overwrite — there's no ambiguity to confirm.
|
||||
# The runtime's `define` blocks are intended to overwrite ? there's no ambiguity to confirm.
|
||||
set confirm off
|
||||
|
||||
# Source the runtime file (re-defines commands with real impls + data).
|
||||
source build/gen/gdb_tape_atoms_runtime.gdb
|
||||
source build/gdb_tape_atoms_runtime.gdb
|
||||
set confirm on
|
||||
echo [gdb_tape_atoms] Runtime sourced successfully (9 commands now have real implementations).
|
||||
|
||||
@@ -7,18 +7,11 @@
|
||||
---
|
||||
--- Ownership: the canonical `ctx.shared.corpus` supplies cross-source registries, while each `src.scan` supplies its source's declarations and bodies.
|
||||
--- A context without `ctx.shared.corpus` is rejected with an explicit canonical-corpus message.
|
||||
---
|
||||
--- Writes `<ctx.out_root>/<dir_basename>.errors.h` once per module, with `#error` directives for findings that the C compile surfaces.
|
||||
--- `passes/report.lua` renders annotations.txt from `corpus.sources_by_dir`, re-validating each source through `M.validate()`.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
|
||||
|
||||
-- Bootstrap follows the entry scripts; `scripts/duffle_paths.lua` sets package.path and package.cpath. See `ps1_meta.lua` for the rationale.
|
||||
-- `debug.getinfo(1, "S").source` locates this file for standalone and orchestrated runs, then `duffle_paths.lua` returns the loaded `duffle` module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local write_file = duffle.write_file
|
||||
local ensure_dir = duffle.ensure_dir
|
||||
|
||||
-- The annotation pass reads the source-derived registries from scan_source:
|
||||
-- * pipe_ctx.register_alias_registry — for atom_dbg_reg_default(R_X, ...) and atom_reg_types(R_X, ...) member-identity checks
|
||||
@@ -605,40 +598,6 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
||||
}
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-DIRECTORY (per-module) output: errors.h + annotations.txt
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Render `<dir_basename>.errors.h` with `#error` directives for every error found across all sources in the directory.
|
||||
--- Empty directories (no errors, no atoms) produce no file.
|
||||
local function emit_module_errors_h(ctx, dir_basename, atoms_count, errors, sources)
|
||||
if atoms_count == 0 and #errors == 0 then
|
||||
return nil
|
||||
end
|
||||
local out_path = ctx.out_root .. "/" .. dir_basename .. ".errors.h"
|
||||
local lines = {
|
||||
"// Auto-generated by ps1_meta.lua (passes/annotation.lua) — DO NOT EDIT",
|
||||
string.format("// Module: %s Sources: %d", dir_basename, #sources),
|
||||
"#pragma once",
|
||||
"",
|
||||
}
|
||||
if #errors == 0 then
|
||||
lines[#lines + 1] = "// annotation pass OK"
|
||||
else
|
||||
for _, e in ipairs(errors) do
|
||||
local src_tag = ""
|
||||
if e.source then
|
||||
local src_name = e.source:match("([^/\\]+)$") or e.source
|
||||
src_tag = src_name .. ": "
|
||||
end
|
||||
lines[#lines + 1] = string.format('#error "%s%s (line %d)"', src_tag, e.msg, e.line)
|
||||
end
|
||||
end
|
||||
ensure_dir(ctx.out_root)
|
||||
write_file(out_path, table.concat(lines, "\n") .. "\n")
|
||||
return out_path
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M.run — orchestrator entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -683,11 +642,6 @@ function M.run(ctx)
|
||||
warnings [#warnings + 1] = { line = w.line, msg = w.msg }
|
||||
end
|
||||
end
|
||||
|
||||
local err_path = emit_module_errors_h(ctx, dir_basename, dir_atoms, dir_errors, dir_sources)
|
||||
if err_path then
|
||||
table.insert(outputs, { errors_h = err_path })
|
||||
end
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
|
||||
@@ -4,20 +4,18 @@
|
||||
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime
|
||||
--- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
|
||||
---
|
||||
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs: one
|
||||
--- `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
||||
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
|
||||
--- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
||||
---
|
||||
--- **Two output forms** (per the workspace's per-emission-form pattern from `guide_metaprogram_ssdl.md`):
|
||||
--- 1. **Sourcemap.txt form** — `<out_root>/<basename>.atoms.sourcemap.txt`. Format-version-tagged for forward-compat.
|
||||
--- Lives in `<out_root>/` (build/gen). Mirrors the convention used by `annotation.lua`
|
||||
--- (`<out_root>/<basename>.errors.h`) and `static_analysis.lua` (`<out_root>/<basename>.static_analysis.txt`).
|
||||
--- Two output forms:
|
||||
--- 1. Markdown form: Handled by `passes/report.lua` (writes `<module>.atoms.md`).
|
||||
--- The render functions `render_source_map` + `render_provenance` are exported for `report.lua` to call directly.
|
||||
--- Compile artifacts (`*.macs.h`, `*.offsets.h`) stay in `<source_dir>/gen/`.
|
||||
--- 2. **gdb-runtime form** — `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`. A pure gdb command script — addresses come
|
||||
--- from `nm`, the 9 user commands are static `define ... end` blocks. Emitted when `ctx.flags.gdb_runtime` is true
|
||||
--- AND `ctx.flags.elf_path` points to an existing ELF. Useful for `gdb-multiarch --without-python` users
|
||||
--- (the common case on Windows MinGW builds) — `source <path>` loads it with no Python / Tcl / Guile required.
|
||||
--- 2. `gdb_tape_atoms_runtime.gdb`: Post-link opt-in (`ctx.flags.gdb_runtime`),
|
||||
--- so the gdb wrapper script + the generated runtime script share the same canonical location.
|
||||
--- Triggered by `--post-link` or `--gdb-runtime`.
|
||||
---
|
||||
--- **Output format** (sourcemap.txt form):
|
||||
--- Output forma (sourcemap.txt form):
|
||||
--- ```
|
||||
--- # FORMAT_VERSION 1
|
||||
--- # auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT
|
||||
@@ -32,8 +30,6 @@
|
||||
--- ```
|
||||
---
|
||||
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
|
||||
---
|
||||
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
@@ -458,7 +454,22 @@ local function emit_gdb_runtime(ctx)
|
||||
-- Confirmation line for the source operator.
|
||||
lines[#lines + 1] = 'printf "[gdb_tape_atoms] runtime loaded %d atoms from %s\\n", $__atom_count, $__elf_path'
|
||||
|
||||
local out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
|
||||
local out_path
|
||||
-- Move out of `<out_root>/gdb_tape_atoms_runtime.gdb` to `<out_root>/../gdb_tape_atoms_runtime.gdb` when the conventional `<out_root>` is `<build>/gen`
|
||||
-- (any equivalent spelling — relative, absolute backslash, absolute forward-slash, trailing-separator variants).
|
||||
-- This puts the gdb runtime alongside the ELF at `build/` rather than under the report subdir.
|
||||
local function ends_with_gen_dir(p)
|
||||
if type(p) ~= "string" then return false end
|
||||
return p:match("[/\\]gen[/\\]?$") ~= nil or p == "build/gen" or p == "build\\gen"
|
||||
end
|
||||
if ends_with_gen_dir(ctx.out_root) then
|
||||
-- Strip the trailing `/gen` segment, then write the runtime script under `build/`.
|
||||
-- e.g. "C:/projects/Pikuma/ps1/build/gen" -> "C:/projects/Pikuma/ps1/build".
|
||||
local parent = ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
|
||||
out_path = parent .. "/gdb_tape_atoms_runtime.gdb"
|
||||
else
|
||||
out_path = ctx.out_root .. "/gdb_tape_atoms_runtime.gdb"
|
||||
end
|
||||
duffle.ensure_dir(duffle.dirname(out_path))
|
||||
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||
-- io.stderr:write(string.format("[atoms_source_map] wrote %s (%d atoms)\n", out_path, #matched))
|
||||
@@ -470,10 +481,62 @@ end
|
||||
|
||||
local M = {}
|
||||
|
||||
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`, emit two files
|
||||
--- in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation). When `ctx.flags.gdb_runtime`
|
||||
--- is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||
-- Expose the pure render functions so `report.lua` and the focused tests can call them directly without triggering the file-emit path.
|
||||
M.render_source_map = render_source_map
|
||||
M.render_provenance = render_provenance
|
||||
|
||||
--- Render ONE atom's sourcemap stanza.
|
||||
--- @param atom table -- atom record (must have `atom.paths` populated)
|
||||
--- @return string
|
||||
function M.render_atom_source_map(atom)
|
||||
assert(type(atom) == "table", "render_atom_source_map: atom must be a table")
|
||||
assert(type(atom.paths) == "table", "render_atom_source_map: atom.paths must be a table")
|
||||
local entries, total = canonical_word_entries(atom)
|
||||
local lines = {}
|
||||
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||
for _, entry in ipairs(entries) do
|
||||
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
|
||||
entry.pos, entry.line, entry.text)
|
||||
end
|
||||
lines[#lines + 1] = "ENDATOM"
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Render ONE atom's provenance stanza — no per-file format header, no enumeration of other atoms.
|
||||
---
|
||||
--- `rel_path` is the source path (forward-slashes) embedded in every `CALL` line.
|
||||
--- The .md caller (report.lua) is expected to derive this once per `## <source>` heading and pass it down for each atom in that source.
|
||||
--- @param atom table -- atom record (must have `atom.paths` populated)
|
||||
--- @param wc table -- identity alias of `corpus.word_counts`
|
||||
--- @param rel_path string -- source path (forward-slashes) for `CALL` fields
|
||||
--- @return string
|
||||
function M.render_atom_provenance(atom, wc, rel_path)
|
||||
assert(type(atom) == "table", "render_atom_provenance: atom must be a table")
|
||||
assert(type(atom.paths) == "table", "render_atom_provenance: atom.paths must be a table")
|
||||
assert(type(rel_path) == "string", "render_atom_provenance: rel_path must be a string")
|
||||
local entries, total = canonical_word_entries(atom)
|
||||
local lines = {}
|
||||
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||
for _, entry in ipairs(entries) do
|
||||
local inv = entry.invocation
|
||||
local macro_count = inv and wc and wc["mac_" .. inv.component_name]
|
||||
if inv and macro_count ~= nil then
|
||||
lines[#lines + 1] = string.format(
|
||||
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
|
||||
entry.pos, rel_path, entry.line, inv.component_name,
|
||||
inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||
else
|
||||
lines[#lines + 1] = string.format(
|
||||
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||
end
|
||||
end
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
|
||||
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
||||
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
@@ -495,34 +558,8 @@ function M.run(ctx)
|
||||
}
|
||||
end
|
||||
|
||||
-- Always emit the text form (per-source).
|
||||
for _, src in ipairs(corpus.source_order) do
|
||||
local has_projection = false
|
||||
for _, atom in ipairs((src.scan or {}).atoms or {}) do
|
||||
if (atom.kind == "atom" or atom.kind == "raw_atom") and atom.paths then
|
||||
has_projection = true; break
|
||||
end
|
||||
end
|
||||
if not has_projection then
|
||||
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
|
||||
if atom.paths then has_projection = true; break end
|
||||
end
|
||||
end
|
||||
if has_projection then
|
||||
local basename = duffle.basename_no_ext(src.path)
|
||||
-- (1) atoms.sourcemap.txt — format-1 per-word call-site map.
|
||||
local sourcemap_path = ctx.out_root .. "/" .. basename .. ".atoms.sourcemap.txt"
|
||||
local sourcemap_body = render_source_map(src)
|
||||
-- (2) atoms.provenance.txt — format-1 per-word definition/body map.
|
||||
local prov_path = ctx.out_root .. "/" .. basename .. ".atoms.provenance.txt"
|
||||
local prov_body = render_provenance(src, wc)
|
||||
duffle.ensure_dir(duffle.dirname(sourcemap_path))
|
||||
duffle.write_file_lf(sourcemap_path, sourcemap_body)
|
||||
duffle.write_file_lf(prov_path, prov_body)
|
||||
outputs[#outputs + 1] = { kind = "report", path = sourcemap_path }
|
||||
outputs[#outputs + 1] = { kind = "report", path = prov_path }
|
||||
end
|
||||
end
|
||||
-- atoms.sourcemap.txt + atoms.provenance.txt content moved to report.lua via `<module>.atoms.md` markdown file.
|
||||
-- This pass emits only the post-link gdb_runtime artifact (see emit_gdb_runtime below).
|
||||
|
||||
-- Optionally emit the gdb-runtime form (post-link, one file per build).
|
||||
if ctx.flags and ctx.flags.gdb_runtime then
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
||||
---
|
||||
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
|
||||
--- then resolves the function-args string from the preceding `FI_ MipsAtom ac_X(...)` declaration via a backward walk.
|
||||
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||
---
|
||||
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||
---
|
||||
@@ -29,7 +29,7 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- Atom component declaration identifiers.
|
||||
local ATOM_COMP_PROC = "MipsAtomComp_Proc_"
|
||||
local MIPS_ATOM = "MipsAtom" -- prefix on the function declaration that wraps an AtomComp_Proc_
|
||||
local MIPS_ATOM = "Slice_MipsCode" -- prefix on the function declaration that wraps an AtomComp_Proc_
|
||||
|
||||
-- Component-name prefixes.
|
||||
local AC_PREFIX = "ac_" -- arg to MipsAtomComp_(ac_X); the X is the atom name
|
||||
@@ -97,9 +97,9 @@ local M = {}
|
||||
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
||||
---
|
||||
--- Convention: function form is
|
||||
--- `FI_ MipsAtom ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
|
||||
--- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
|
||||
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
|
||||
--- We then verify the preceding context ends with `MipsAtom`
|
||||
--- We then verify the preceding context ends with `Slice_MipsCode`
|
||||
--- (the function-decl keyword with possible qualifiers between).
|
||||
---
|
||||
--- @param source string
|
||||
|
||||
@@ -71,9 +71,15 @@ local DW_LNE_set_address = DWARF_LINE_OPS.DW_LNE_set_address
|
||||
local DW_RLE_end_of_list = DWARF5_RNGLISTS.end_of_list
|
||||
local DW_RLE_start_length = DWARF5_RNGLISTS.start_length
|
||||
|
||||
-- File index 11 in the existing main line unit is hello_gte_tape.c.
|
||||
-- The injector extends that unit rather than appending an unreferenced unit.
|
||||
local ATOM_SOURCE_FILE_INDEX = 11
|
||||
-- File-index lookup for the existing main line unit (Unit 2).
|
||||
-- Populated at pass start by `init_file_index_lookup(elf_path)` from the runtime ELF (see `elf_dwarf.read_line_unit_file_table`).
|
||||
-- The hardcoded indices and the `PROVENANCE_BASENAME_TO_FILE_INDEX` table that previously lived here were retired in `conductor/tracks/dwarf_file_index_lookup_20260731/`
|
||||
-- (red of the
|
||||
-- `TODO(Ed): Remove this HARDCODE` from line 156); the runtime lookup reads the
|
||||
-- actual gcc-emitted `.debug_line` file table instead.
|
||||
local _file_index_by_basename = nil -- [basename] = 1-based line-table file index
|
||||
local _file_path_by_index = nil -- [1-based index] = full source path (diagnostics / future consumers)
|
||||
local _default_atom_source_index = nil -- any valid index used in opaque-row fallbacks
|
||||
|
||||
-- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31
|
||||
-- (see collect_per_source_registries + by_alias in build_inserted_children).
|
||||
@@ -98,9 +104,8 @@ local ABBREV_INLINED_SUBROUTINE = 0x6C -- 108: DW_TAG_inlined_subroutine with
|
||||
-- (each field transitions from tape memory to GPR at load_pc + 8 = MIPS I load-delay slot boundary).
|
||||
local ABBREV_BIND_VAR_LOCLIST = 0x6D -- 109: DW_TAG_variable no children + DW_AT_type = ref4 + DW_AT_location = sec_offset
|
||||
-- Typed-view pointer_type (for the synthetic V4_S2* / V3_S2* / U4* / void* chains).
|
||||
-- MUST be a fresh abbrev code in the appended table — emitting uleb128(9) collides with GCC's
|
||||
-- existing abbrev 9 (a pointer_type that carries DW_AT_byte_size + DW_AT_type), so gdb misparses
|
||||
-- our 4-byte ref4 as (byte_size, type[0..2]) and lands the cursor mid-attribute.
|
||||
-- MUST be a fresh abbrev code in the appended table — emitting uleb128(9) collides with GCC's existing abbrev 9
|
||||
-- (a pointer_type that carries DW_AT_byte_size + DW_AT_type), so gdb misparses our 4-byte ref4 as (byte_size, type[0..2]) and lands the cursor mid-attribute.
|
||||
local ABBREV_TYPED_VIEW_POINTER = 0x6E -- 110: DW_TAG_pointer_type no children + DW_AT_type = ref4 (typed-view / U4 / void chain)
|
||||
|
||||
-- DWARF5 §7.7.3 loclist opcodes.
|
||||
@@ -153,44 +158,66 @@ local DW_AT_inline = 0x20 -- DWARF5 §7.7.1: DW_AT_inline (used by a
|
||||
local DW_AT_decl_file = 0x3A -- DWARF5 §7.7.1: DW_AT_decl_file (1-based file index into the CU's file table)
|
||||
local DW_AT_decl_line = 0x3B -- DWARF5 §7.7.1: DW_AT_decl_line
|
||||
|
||||
-- File index lookup table for the existing main line unit (Unit 2).
|
||||
-- Provenance paths come back with mixed slashes; we normalize to basename and look up against the line unit's actual file table.
|
||||
-- Current scope has two provenance basenames: hello_gte_tape.c (the atom's call site) and lottes_tape.h (the component definition).
|
||||
-- Both live in the existing gcc-generated line unit; their 1-based indices are stable across rebuilds because the include order
|
||||
-- in code/gte_hello/hello_gte.c determines the unit's file table.
|
||||
-- gcc only adds a file to the line table when it has actual line-number entries;
|
||||
-- headers that are pure macros/typedefs (dsl.h, memory.h, math.h, mips.h, gp.h, gte.h, etc.) never appear.
|
||||
-- lottes_tape.h is the FIRST include that emits line entries (MipsAtomComp_ declarations), so it is the FIRST entry after the primary file.
|
||||
local PROVENANCE_BASENAME_TO_FILE_INDEX = {
|
||||
["hello_gte_tape.c"] = ATOM_SOURCE_FILE_INDEX, -- = 11
|
||||
["lottes_tape.h"] = 2,
|
||||
}
|
||||
-- Replaced the hardcoded `ATOM_SOURCE_FILE_INDEX = 11` and the `PROVENANCE_BASENAME_TO_FILE_INDEX` table below with a runtime lookup
|
||||
-- (`init_file_index_lookup` + `resolve_provenance_file_index`) that reads the actual `.debug_line` file table from the post-link ELF.
|
||||
|
||||
--- Populate the module-level file-index lookup table from the `.debug_line` section of the post-link ELF pointed at by `elf_path`.
|
||||
--- This MUST be called exactly once at pass start (from `M.run`) before any `resolve_provenance_file_index` invocation;
|
||||
--- downstream callers handle a nil table as "no file info available; fall back to errors".
|
||||
---
|
||||
--- The lookup uses `elf_dwarf.read_line_unit_file_table` (which parses both DWARF3 and DWARF5 line-program units —
|
||||
--- the crt0.s assembler-side DWARF5 unit may emit non-standard form codes for paths and is intentionally skipped).
|
||||
--- @param elf_path string|nil
|
||||
local function init_file_index_lookup(elf_path)
|
||||
if not elf_path or elf_path == "" then return end
|
||||
local b2i, _basenames, paths = elf_dwarf.read_line_unit_file_table(elf_path)
|
||||
if type(b2i) ~= "table" or type(paths) ~= "table" then
|
||||
io.stderr:write("[dwarf_injection] read_line_unit_file_table returned no file table for: " .. tostring(elf_path) .. "\n")
|
||||
return
|
||||
end
|
||||
_file_index_by_basename = b2i
|
||||
_file_path_by_index = paths
|
||||
-- Pick any valid index for the opaque-row fallbacks at lines 466 + 570
|
||||
-- (both sites legitimately want "any file index"; gdb resolves whatever index we emit to whatever that file's line happens to be).
|
||||
for idx in pairs(paths) do
|
||||
_default_atom_source_index = idx
|
||||
break
|
||||
end
|
||||
end
|
||||
|
||||
--- Resolve an absolute provenance path to the line-unit file index used by the emitting line program.
|
||||
--- Normalizes mixed `/` and `\` separators to a basename and looks it up against the known file table.
|
||||
--- Normalizes mixed `/` and `\` separators to a basename and looks it up against the runtime-computed file table populated by `init_file_index_lookup`.
|
||||
---
|
||||
--- Fails loudly on an unknown provenance basename: adding a new component source file requires extending
|
||||
--- `PROVENANCE_BASENAME_TO_FILE_INDEX` so the line-program emission contract stays explicit.
|
||||
--- Silent fallback to ATOM_SOURCE_FILE_INDEX would mask the new-file case by misattributing component rows to the atom's source file.
|
||||
--- @param path string -- absolute provenance path (e.g. "C:/.../lottes_thttps://www.youtube.com/watch?v=ORM4yLkdKx8ape.h" or "C:\\...\\lottes_tape.h")
|
||||
--- Fails loudly on an unknown provenance basename: adding a new component source file will produce a clear error message naming the missing basename and listing the .debug_line file table contents,
|
||||
--- so the user can either confirm the gcc include order, the unity-root, or the `.debug_line` file table contents.
|
||||
--- Silent fallback would mask the new-file case by misattributing component rows to an arbitrary source file.
|
||||
--- @param path string -- absolute provenance path (mixed slashes accepted)
|
||||
--- @return integer -- 1-based line-unit file index
|
||||
local function resolve_provenance_file_index(path)
|
||||
if _file_index_by_basename == nil then
|
||||
error("[dwarf_injection] resolve_provenance_file_index called before init_file_index_lookup. "
|
||||
.. "Is M.run being entered correctly (with --elf)?")
|
||||
end
|
||||
if path == nil or path == "" then
|
||||
error("[dwarf_injection] resolve_provenance_file_index: empty path")
|
||||
end
|
||||
-- Normalize backslashes → forward slashes (paths arrive with mixed separators from the provenance file: forward slashes from Lua's io.lines;
|
||||
-- backslashes if the input ever round-trips through Windows shell expansion).
|
||||
-- Normalize backslashes → forward slashes (paths arrive with mixed separators from the provenance file).
|
||||
local normalized = path:gsub("\\", "/")
|
||||
-- Take the last path component (the basename).
|
||||
local basename = normalized:match("([^/]+)$") or normalized
|
||||
local idx = PROVENANCE_BASENAME_TO_FILE_INDEX[basename]
|
||||
if idx == nil then
|
||||
error(string.format(
|
||||
"[dwarf_injection] resolve_provenance_file_index: unknown provenance basename '%s' (from '%s'). "
|
||||
.. "Extend PROVENANCE_BASENAME_TO_FILE_INDEX in passes/dwarf_injection.lua.",
|
||||
basename, path))
|
||||
local idx = _file_index_by_basename[basename]
|
||||
if idx ~= nil then return idx end
|
||||
-- Last-resort exact-path match (handles paths that don't reduce to a known basename).
|
||||
for i, p in pairs(_file_path_by_index) do
|
||||
if p and p:gsub("\\", "/") == normalized then return i end
|
||||
end
|
||||
return idx
|
||||
-- Build an error message listing the known basenames for fast diagnostics.
|
||||
local known = {}
|
||||
for k in pairs(_file_index_by_basename) do known[#known + 1] = k end
|
||||
table.sort(known)
|
||||
error(string.format("[dwarf_injection] resolve_provenance_file_index: unknown provenance basename '%s' (from '%s'). "
|
||||
.. "Known basenames in the .debug_line file table (%d): %s"
|
||||
, basename, path, #known, table.concat(known, ", ")))
|
||||
end
|
||||
|
||||
local DW_FORM_addr = 0x01
|
||||
@@ -461,7 +488,7 @@ local function build_atom_sequence(atom)
|
||||
if atom.debug_skip then
|
||||
return table.concat({
|
||||
set_address(atom.addr),
|
||||
set_file(ATOM_SOURCE_FILE_INDEX),
|
||||
set_file(resolve_provenance_file_index(atom.src_path)),
|
||||
advance_line(atom.entries[1].line - 1),
|
||||
negate_stmt(),
|
||||
copy_op(),
|
||||
@@ -508,10 +535,9 @@ local function build_atom_sequence(atom)
|
||||
-- * If the invocation's body has any NESTED invocations (parent_id == top_inv.id), the body's
|
||||
-- first content is the call_line of the earliest nested invocation (by start_pos).
|
||||
-- * Otherwise (only RAW words in the body), it's the line of the first raw word = body_lines[1].
|
||||
-- This is the value the multi-row PC's body_lines[1] row must reference for source-order display:
|
||||
-- `anc.body_lines[1]` is the line of the FIRST WORD (which for an outer whose body starts with a
|
||||
-- nested expansion is inside the inner's expansion = wrong for display purposes); `anc.body_first_line`
|
||||
-- is the body's first content line in the parent's source (= correct for display).
|
||||
-- This is the value the multi-row PC's body_lines[1] row must reference for source-order display: `anc.body_lines[1]` is the line of the FIRST WORD
|
||||
-- (which for an outer whose body starts with a nested expansion is inside the inner's expansion = wrong for display purposes);
|
||||
-- `anc.body_first_line` is the body's first content line in the parent's source (= correct for display).
|
||||
local body_first_line_of = {}
|
||||
for _, top_inv in ipairs(invs) do
|
||||
local earliest_nested_call_line = nil
|
||||
@@ -565,7 +591,7 @@ local function build_atom_sequence(atom)
|
||||
parts[#parts + 1] = copy_op()
|
||||
end
|
||||
|
||||
local call_file_idx = ATOM_SOURCE_FILE_INDEX
|
||||
local call_file_idx = resolve_provenance_file_index(atom.src_path)
|
||||
|
||||
-- --- Atom entry (idx 1) -------------------------------------------------
|
||||
local entry_1 = atom.entries[1]
|
||||
@@ -574,13 +600,11 @@ local function build_atom_sequence(atom)
|
||||
-- If atom entry 1 starts inside an invocation, walk the ancestry and emit a call-site row + (when applicable)
|
||||
-- a body_lines[1] row for every active ancestor. For a non-nested invocation this is just the one pair;
|
||||
-- for nested invocations this emits the outer call-site + body_lines[1] rows BEFORE the inner pair so the debugger displays
|
||||
-- the outer body line at the inner's first word
|
||||
-- (PROBLEM B fix).
|
||||
-- the outer body line at the inner's first word (PROBLEM B fix).
|
||||
--
|
||||
-- A marked OUTERMOST ancestor's body_lines[1] row is suppressed at this PC (the existing full-skip
|
||||
-- contract is preserved for the marked outer range); its call-site row IS still emitted as a
|
||||
-- statement. Marked INNER ancestors always emit their body_lines[1] row with is_stmt=false
|
||||
-- (the per-invocation `want_body = not inv.debug_skip` predicate).
|
||||
-- A marked OUTERMOST ancestor's body_lines[1] row is suppressed at this PC (the existing full-skip contract is preserved for the marked outer range);
|
||||
-- Its call-site row IS still emitted as a statement.
|
||||
-- Marked INNER ancestors always emit their body_lines[1] row with is_stmt=false (the per-invocation `want_body = not inv.debug_skip` predicate).
|
||||
if #entry_1_ancestry == 0 then
|
||||
-- RAW word at atom entry: single call-site row, always a statement target.
|
||||
emit_row(call_file_idx, entry_1.line, true)
|
||||
@@ -588,9 +612,8 @@ local function build_atom_sequence(atom)
|
||||
-- Atom starts in an invocation. Walk the ancestry outermost-first.
|
||||
-- Each ancestor emits one call-site row (statement) and one body_lines[1] row
|
||||
-- (statement iff unmarked; suppressed for marked outermost).
|
||||
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content
|
||||
-- line in the parent's source), NOT anc.body_lines[1] (= the line of the first WORD,
|
||||
-- which is wrong when the outer's body starts with a nested call).
|
||||
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
|
||||
-- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call).
|
||||
for ai, anc in ipairs(entry_1_ancestry) do
|
||||
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
|
||||
assert(anc.body_lines[1] ~= nil
|
||||
@@ -615,20 +638,18 @@ local function build_atom_sequence(atom)
|
||||
|
||||
if inv and idx == inv.start_pos + 1 then
|
||||
-- First word of the innermost active invocation (PROBLEM B fix — nested-display rule).
|
||||
-- Walk the active ancestry outermost-first; for each ancestor emit a call-site row
|
||||
-- (statement) + a body_lines[1] row. The inner-most invocation's call-site + body pair
|
||||
-- become the LAST two rows in the sequence. Marked outermost ancestors suppress their
|
||||
-- body_lines[1] row at this PC (the existing full-skip contract is preserved for the
|
||||
-- marked outer range); all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
|
||||
-- Walk the active ancestry outermost-first; for each ancestor emit a call-site row (statement) + a body_lines[1] row.
|
||||
-- The inner-most invocation's call-site + body pair become the LAST two rows in the sequence.
|
||||
-- Marked outermost ancestors suppress their body_lines[1] row at this PC (the existing full-skip contract is preserved for the marked outer range);
|
||||
-- all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
|
||||
--
|
||||
-- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC
|
||||
-- for debugger context: source-level stepping now shows the outer body line (not the
|
||||
-- inner body line) when stepping into the inner. PROBLEM B fix.
|
||||
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content
|
||||
-- line in the parent's source), NOT anc.body_lines[1] (= the line of the first WORD,
|
||||
-- which is wrong when the outer's body starts with a nested call: gdb 12.1 picks the
|
||||
-- displayed line as the LAST row at the same PC in byte-stream order, so the disc=1 row's
|
||||
-- value matters for what's shown when stepping into the nested case).
|
||||
-- for debugger context: source-level stepping now shows the outer body line
|
||||
-- (not the inner body line) when stepping into the inner. PROBLEM B fix.
|
||||
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
|
||||
-- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call:
|
||||
-- gdb 12.1 picks the displayed line as the LAST row at the same PC in byte-stream order,
|
||||
-- so the disc=1 row's value matters for what's shown when stepping into the nested case).
|
||||
local ancestry = ancestry_idx[idx]
|
||||
for ai, anc in ipairs(ancestry) do
|
||||
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
|
||||
@@ -647,10 +668,9 @@ local function build_atom_sequence(atom)
|
||||
-- Subsequent body word of the innermost active invocation: `body_lines[k]` is indexed by the 1-based offset of this word inside the invocation.
|
||||
-- Both `idx` (1-based DWARF entry index) and `inv.start_pos` (0-based emitted-word position stamped at `emit_invoke_begin`) come from the same
|
||||
-- monotonic counter, so `idx - inv.start_pos` is exactly the 1-based k (the first word of the invocation has `idx == inv.start_pos + 1`, hence `k == 1`).
|
||||
-- atom_dbg_step_ux_20260725: `want_body = not inv.debug_skip`. The previous `want = not marked_idx[idx]`
|
||||
-- (which suppressed ALL body rows when any ancestor was marked) is replaced by the per-invocation
|
||||
-- predicate. Marked invocations emit non-statement body rows at every body word; unmarked
|
||||
-- invocations emit statement body rows.
|
||||
-- atom_dbg_step_ux_20260725: `want_body = not inv.debug_skip`.
|
||||
-- The previous `want = not marked_idx[idx]` (which suppressed ALL body rows when any ancestor was marked) is replaced by the per-invocation predicate.
|
||||
-- Marked invocations emit non-statement body rows at every body word; unmarked invocations emit statement body rows.
|
||||
assert(inv.body_lines, "missing body_lines: emitter did not run emission-model")
|
||||
local words_into = idx - inv.start_pos
|
||||
assert(inv.body_lines[words_into] ~= nil
|
||||
@@ -704,7 +724,9 @@ local function build_atom_table(corpus, addrs)
|
||||
local atoms_by_name = corpus.atoms_by_name or {}
|
||||
|
||||
-- Per-atom ingest. Returns nil if the atom is absent from the corpus; the caller skips it via the `if atom then ...` guard.
|
||||
local function ingest_atom(name, info)
|
||||
-- `src_path` is the absolute source path that declared this atom; the build_atom_table iteration below threads `src.path` through.
|
||||
-- This is consumed by `build_atom_sequence::set_file(...)` for opaque-row fallbacks + raw-word rows (atoms where no invocation ancestry exists).
|
||||
local function ingest_atom(name, info, src_path)
|
||||
local atom_record = atoms_by_name[name]
|
||||
if not atom_record then return nil end
|
||||
|
||||
@@ -730,6 +752,7 @@ local function build_atom_table(corpus, addrs)
|
||||
words = #word_events,
|
||||
entries = entries,
|
||||
debug_skip = atom_record.debug_skip == true,
|
||||
src_path = src_path or "",
|
||||
}
|
||||
|
||||
-- Consume invocation records from `atom.paths.invocations`. It is the single producer of per-invocation body_lines, per-invocation debug_skip,
|
||||
@@ -753,9 +776,26 @@ local function build_atom_table(corpus, addrs)
|
||||
end
|
||||
|
||||
local out = {}
|
||||
for name, info in pairs(addrs) do
|
||||
local atom = ingest_atom(name, info)
|
||||
if atom then out[#out + 1] = atom end
|
||||
-- Walk every source's atom list (which preserves source order + per-source src_path).
|
||||
-- Cross-ref with the nm symbol table; atoms absent from `addrs` are skipped (an atom
|
||||
-- declared in source but not emitted as a symbol is a metaprogram or atom-info bug, not
|
||||
-- a source-correlation bug — emit_no_emit would catch it upstream).
|
||||
for _, src in ipairs((corpus and corpus.source_order) or {}) do
|
||||
local src_path = src.path or ""
|
||||
for _, atom_rec in ipairs(((src.scan or {}).atoms) or {}) do
|
||||
local info = addrs[atom_rec.name or atom_rec.raw_name]
|
||||
if info then
|
||||
local atom = ingest_atom(atom_rec.name or atom_rec.raw_name, info, src_path)
|
||||
if atom then out[#out + 1] = atom end
|
||||
end
|
||||
end
|
||||
for _, atom_rec in ipairs(((src.scan or {}).raw_atoms) or {}) do
|
||||
local info = addrs[atom_rec.name or atom_rec.raw_name]
|
||||
if info then
|
||||
local atom = ingest_atom(atom_rec.name or atom_rec.raw_name, info, src_path)
|
||||
if atom then out[#out + 1] = atom end
|
||||
end
|
||||
end
|
||||
end
|
||||
table.sort(out, function(a, b) return a.addr < b.addr end)
|
||||
return out
|
||||
@@ -2186,6 +2226,10 @@ function M.run(ctx)
|
||||
-- reading them just returns "" which is the "missing" case the builder handles.
|
||||
".debug_loc", ".debug_loclists",
|
||||
})
|
||||
-- Resolve the per-file line-table indices from the same .debug_line bytes;
|
||||
-- this MUST run before any atom sequence is emitted (build_atom_sequence below
|
||||
-- calls resolve_provenance_file_index when populating call-site / body rows).
|
||||
init_file_index_lookup(elf_path)
|
||||
-- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation).
|
||||
-- `corpus` is the sole canonical source projection.
|
||||
local corpus = (ctx.shared and ctx.shared.corpus) or {}
|
||||
|
||||
+410
-287
@@ -7,9 +7,6 @@
|
||||
---
|
||||
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
|
||||
--- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
@@ -18,17 +15,21 @@
|
||||
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
||||
-- Bootstrap: see `ps1_meta.lua` for the rationale.
|
||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
||||
-- both standalone and when require'd from the orchestrator.
|
||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
|
||||
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
|
||||
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
|
||||
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
|
||||
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
|
||||
|
||||
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
|
||||
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
|
||||
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
|
||||
local atoms_source_map = dofile(_bootstrap_dir .. "atoms_source_map.lua")
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Constants
|
||||
@@ -44,8 +45,7 @@ local SECTION_HEADER_MACROS = "── Macro word-count declarations ───
|
||||
local SECTION_HEADER_ERRORS = "── Errors ──────────────────────────────────────────────"
|
||||
local SECTION_HEADER_WARNINGS = "── Warnings ────────────────────────────────────────────"
|
||||
|
||||
-- Lua pattern that captures the basename (last path segment) of a
|
||||
-- forward- or back-slash separated path.
|
||||
-- Lua pattern that captures the basename (last path segment) of a forward- or back-slash separated path.
|
||||
local BASENAME_PATTERN = "([^/\\]+)$"
|
||||
|
||||
-- Debug flag name — set to truthy in `_G` to enable verbose logging.
|
||||
@@ -148,328 +148,451 @@ local function source_basename(path)
|
||||
return path:match(BASENAME_PATTERN) or path
|
||||
end
|
||||
|
||||
--- (internal) Format a single annotation entry as one rendered line.
|
||||
--- @param a AnnotEntry
|
||||
--- @param src_name string
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Markdown renderers (consolidated-report-files refactor, 2026-07-26)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Render the thin project-wide summary (`build/atom_meta_report.summary.md`).
|
||||
--- @param all_results { module:string, atoms:integer, annots:integer, binds:integer,
|
||||
--- macros:integer, findings:integer, errors:integer,
|
||||
--- warnings:integer, info:integer }[]
|
||||
--- @return string
|
||||
local function format_annot_line(a, src_name)
|
||||
if a.error then
|
||||
return string.format(" ✗ line %d %s [ERROR: %s] [%s]", a.line, a.macro or "?", a.error, src_name)
|
||||
end
|
||||
local line = string.format(" ● line %d %s [%s]", a.line, a.name, src_name)
|
||||
if a.binds then line = line .. " binds=" .. a.binds end
|
||||
if #a.reads > 0 then line = line .. " reads={" .. table.concat(a.reads, ",") .. "}" end
|
||||
if #a.writes > 0 then line = line .. " writes={" .. table.concat(a.writes, ",") .. "}" end
|
||||
return line
|
||||
end
|
||||
|
||||
--- (internal) Tally totals across all results in a module.
|
||||
--- @param results AnnotationResult[]
|
||||
--- @return integer, integer, integer, integer, integer, integer
|
||||
local function tally_module_totals(results)
|
||||
local total_atoms, total_annots, total_binds, total_macros = 0, 0, 0, 0
|
||||
local total_errors, total_warnings = 0, 0
|
||||
for _, r in ipairs(results) do
|
||||
total_atoms = total_atoms + #r.atoms
|
||||
total_annots = total_annots + #r.annots
|
||||
total_binds = total_binds + #r.binds
|
||||
total_macros = total_macros + #r.macros
|
||||
total_errors = total_errors + #r.errors
|
||||
total_warnings = total_warnings + #r.warnings
|
||||
end
|
||||
return total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source atom declarations.
|
||||
local function render_module_atoms_section(add, results)
|
||||
add(SECTION_HEADER_ATOMS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, a in ipairs(r.atoms) do
|
||||
add(string.format(" MipsAtom_(%s) line %d [%s]", a.name, a.line, src_name))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source annotation entries.
|
||||
local function render_module_annots_section(add, results)
|
||||
add(SECTION_HEADER_ANNOTS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, a in ipairs(r.annots) do
|
||||
add(format_annot_line(a, src_name))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source Binds_* struct declarations.
|
||||
local function render_module_binds_section(add, results)
|
||||
add(SECTION_HEADER_BINDS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, b in ipairs(r.binds) do
|
||||
add(string.format(" %s line %d %d bytes [%s]", b.name, b.line, b.bytes, src_name))
|
||||
for _, f in ipairs(b.fields) do
|
||||
add(string.format(" +%2d: %s", f.offset, f.name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source macro word-count declarations.
|
||||
local function render_module_macros_section(add, results)
|
||||
add(SECTION_HEADER_MACROS)
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, m in ipairs(r.macros) do
|
||||
add(string.format(" %s line %d words=%d [%s]", m.name, m.line, m.words, src_name))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source errors (one-line + "(none)" if empty).
|
||||
local function render_module_errors_section(add, results, total_errors)
|
||||
add(SECTION_HEADER_ERRORS)
|
||||
if total_errors == 0 then
|
||||
add(" (none)")
|
||||
else
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, e in ipairs(r.errors) do
|
||||
add(string.format(" ✗ line %d %s [%s]", e.line, e.msg, src_name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- (internal) Section renderer: per-source warnings (one-line + "(none)" if empty).
|
||||
local function render_module_warnings_section(add, results, total_warnings)
|
||||
add(SECTION_HEADER_WARNINGS)
|
||||
if total_warnings == 0 then
|
||||
add(" (none)")
|
||||
else
|
||||
for _, r in ipairs(results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, w in ipairs(r.warnings) do
|
||||
add(string.format(" ⚠ line %d %s [%s]", w.line, w.msg, src_name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- SECTION_RENDERERS — data-driven section dispatch (the plex pattern)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
--
|
||||
-- Each entry maps a section to its (header, render_fn). The render_fn signature:
|
||||
-- render_fn(add, results, totals)
|
||||
-- add -- the `add(line)` closure from the surrounding report renderer
|
||||
-- results -- AnnotationResult[] (per-source results)
|
||||
-- totals -- {atoms, annots, binds, macros, errors, warnings} counts
|
||||
--
|
||||
-- Sections that need to render "(none)" vs iterate use totals.errors / totals.warnings;
|
||||
-- other sections ignore the totals arg.
|
||||
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
|
||||
local SECTION_RENDERERS = {
|
||||
{ header = SECTION_HEADER_ATOMS, render = render_module_atoms_section },
|
||||
{ header = SECTION_HEADER_ANNOTS, render = render_module_annots_section },
|
||||
{ header = SECTION_HEADER_BINDS, render = render_module_binds_section },
|
||||
{ header = SECTION_HEADER_MACROS, render = render_module_macros_section },
|
||||
{ header = SECTION_HEADER_ERRORS, render = function(add, results, totals) return render_module_errors_section(add, results, totals.errors) end },
|
||||
{ header = SECTION_HEADER_WARNINGS, render = function(add, results, totals) return render_module_warnings_section(add, results, totals.warnings) end },
|
||||
}
|
||||
|
||||
--- Render the per-MODULE annotation report (one `<dir_basename>.annotations.txt`).
|
||||
--- @param dir string -- module directory path
|
||||
--- @param sources SourceFile[] -- sources in this module
|
||||
--- @param results AnnotationResult[] -- per-source validate() results
|
||||
--- @return string -- the rendered report text
|
||||
local function render_module_report(dir, sources, results)
|
||||
local lines = {}
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
add(RULE_THICK)
|
||||
add("ANNOTATION PASS — module " .. source_basename(dir))
|
||||
add(RULE_THICK)
|
||||
add(string.format("Sources: %d", #sources))
|
||||
for _, s in ipairs(sources) do add(" " .. s.path) end
|
||||
add("")
|
||||
|
||||
local total_atoms, total_annots, total_binds, total_macros, total_errors, total_warnings = tally_module_totals(results)
|
||||
add(string.format("Atoms: %d Annotations: %d Binds structs: %d Macro decls: %d",
|
||||
total_atoms, total_annots, total_binds, total_macros))
|
||||
add("")
|
||||
|
||||
-- Bundle the totals so the section renderers don't need separate parameter lists.
|
||||
-- Errors/warnings sections need their total count to decide "(none)" vs iterate.
|
||||
-- Sections without totals (atoms/annots/binds/macros) ignore this arg.
|
||||
local totals = {
|
||||
atoms = total_atoms, annots = total_annots, binds = total_binds,
|
||||
macros = total_macros, errors = total_errors, warnings = total_warnings,
|
||||
local function render_project_summary(all_results)
|
||||
local lines = {
|
||||
"# Project summary",
|
||||
"> Auto-generated by ps1_meta.lua (passes/report.lua).",
|
||||
"",
|
||||
"| module | atoms | annots | binds | macros | findings | errors | warnings | info |",
|
||||
"|--------|-------|--------|-------|--------|----------|--------|----------|------|",
|
||||
}
|
||||
|
||||
-- THE per-section dispatch. ONE loop over SECTION_RENDERERS.
|
||||
-- Each renderer writes its header + content via the `add` closure (pre-bound above).
|
||||
-- Adding a new section = 1 row here + 1 render_<thing>_section function.
|
||||
for _, section in ipairs(SECTION_RENDERERS) do
|
||||
section.render(add, results, totals)
|
||||
local totals = { atoms = 0, annots = 0, binds = 0, macros = 0,
|
||||
findings = 0, errors = 0, warnings = 0, info = 0 }
|
||||
for _, e in ipairs(all_results) do
|
||||
lines[#lines + 1] = string.format(
|
||||
"| %s | %d | %d | %d | %d | %d | %d | %d | %d |",
|
||||
e.module, e.atoms, e.annots, e.binds, e.macros,
|
||||
e.findings, e.errors, e.warnings, e.info)
|
||||
totals.atoms = totals.atoms + e.atoms
|
||||
totals.annots = totals.annots + e.annots
|
||||
totals.binds = totals.binds + e.binds
|
||||
totals.macros = totals.macros + e.macros
|
||||
totals.findings = totals.findings + e.findings
|
||||
totals.errors = totals.errors + e.errors
|
||||
totals.warnings = totals.warnings + e.warnings
|
||||
totals.info = totals.info + e.info
|
||||
end
|
||||
|
||||
lines[#lines + 1] = string.format(
|
||||
"| **TOTAL** | %d | %d | %d | %d | %d | %d | %d | %d |",
|
||||
totals.atoms, totals.annots, totals.binds, totals.macros,
|
||||
totals.findings, totals.errors, totals.warnings, totals.info)
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-project summary
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Render the per-project summary (`build/gen/annotation_validation.txt`).
|
||||
--- Aggregates totals across all sources; lists per-source error counts if any source has errors.
|
||||
--- @param all_results AnnotationResult[]
|
||||
--- Render the per-module verbose source-map markdown (`build/<module>.atoms.md`).
|
||||
--- Per-source sub-section, per-atom stanza with sourcemap + provenance rows.
|
||||
--- Pulls sourcemap + provenance from `atoms_source_map` (no second source walk).
|
||||
--- @param dir string
|
||||
--- @param dir_sources SourceFile[]
|
||||
--- @param wc table<string, integer>
|
||||
--- @return string
|
||||
local function render_project_report(all_results)
|
||||
local lines = {}
|
||||
local function render_module_atoms_md(dir, dir_sources, wc)
|
||||
local dir_basename = source_basename(dir)
|
||||
local lines = {
|
||||
"# " .. dir_basename .. " — atoms (verbose source map)",
|
||||
"> Per-word call-site + provenance. Auto-generated.",
|
||||
"",
|
||||
}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
local src_name = source_basename(src.path)
|
||||
lines[#lines + 1] = "## " .. src_name
|
||||
lines[#lines + 1] = ""
|
||||
-- For each atom with a projection, render its sourcemap + provenance.
|
||||
local atoms_list = {}
|
||||
for _, atom in ipairs((src.scan or {}).atoms or {}) do
|
||||
if atom.paths then atoms_list[#atoms_list + 1] = atom end
|
||||
end
|
||||
for _, atom in ipairs((src.scan or {}).raw_atoms or {}) do
|
||||
if atom.paths then atoms_list[#atoms_list + 1] = atom end
|
||||
end
|
||||
if #atoms_list == 0 then
|
||||
lines[#lines + 1] = "_(no atom projections)_"
|
||||
lines[#lines + 1] = ""
|
||||
else
|
||||
-- Per-source forward-slash path (same one `emit_atom_stanza` / `emit_provenance_stanza` would derive;
|
||||
-- computed once per `## <source>` heading and reused by each atom's `WORD N CALL ...` field).
|
||||
local rel_path = src.path:gsub("\\\\", "/")
|
||||
for _, atom in ipairs(atoms_list) do
|
||||
lines[#lines + 1] = string.format(
|
||||
"### atom: %s (line %d, %d words)",
|
||||
atom.name, atom.line or 0, #(atom.paths.items or {}))
|
||||
lines[#lines + 1] = ""
|
||||
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
|
||||
lines[#lines + 1] = "```"
|
||||
-- Per-atom invariant: call the per-atom renderers, NOT the per-source ones.
|
||||
-- The per-source renderers enumerate every atom in `src`;
|
||||
-- calling them in a per-atom loop would repeat the whole source under every `### atom:` heading.
|
||||
lines[#lines + 1] = atoms_source_map.render_atom_source_map(atom):gsub("\n+$", "")
|
||||
lines[#lines + 1] = "```"
|
||||
lines[#lines + 1] = ""
|
||||
lines[#lines + 1] = "**Provenance** — per-word definition + body:"
|
||||
lines[#lines + 1] = "```"
|
||||
lines[#lines + 1] = atoms_source_map.render_atom_provenance(atom, wc, rel_path):gsub("\n+$", "")
|
||||
lines[#lines + 1] = "```"
|
||||
lines[#lines + 1] = ""
|
||||
end
|
||||
end
|
||||
end
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
|
||||
--- Aggregates annotation + static-analysis content across all sources in `dir`.
|
||||
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
|
||||
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
|
||||
--- @param dir string
|
||||
--- @param dir_sources SourceFile[]
|
||||
--- @param annot_results AnnotationResult[]
|
||||
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
|
||||
--- @return string
|
||||
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
||||
local dir_basename = source_basename(dir)
|
||||
local lines = {
|
||||
"# " .. dir_basename .. " — atom meta report",
|
||||
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
||||
"",
|
||||
}
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
local total_atoms, total_annots, total_macros, total_binds = 0, 0, 0, 0
|
||||
local total_errors, total_warnings = 0, 0
|
||||
for _, r in ipairs(all_results) do
|
||||
total_atoms = total_atoms + #r.atoms
|
||||
total_annots = total_annots + #r.annots
|
||||
total_macros = total_macros + #r.macros
|
||||
total_binds = total_binds + #r.binds
|
||||
total_errors = total_errors + #r.errors
|
||||
total_warnings = total_warnings + #r.warnings
|
||||
-- Module summary table.
|
||||
local n_atoms = 0
|
||||
local n_annot = 0
|
||||
local n_binds = 0
|
||||
local n_macros = 0
|
||||
local n_bare, n_proc = 0, 0
|
||||
for _, r in ipairs(annot_results) do
|
||||
n_atoms = n_atoms + #r.atoms
|
||||
n_annot = n_annot + #r.annots
|
||||
n_binds = n_binds + #r.binds
|
||||
n_macros = n_macros + #r.macros
|
||||
end
|
||||
for _, a in ipairs(sa_results.atoms or {}) do
|
||||
if a.kind == "comp_bare" then n_bare = n_bare + 1
|
||||
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
|
||||
end
|
||||
end
|
||||
|
||||
add(RULE_THICK)
|
||||
add("ANNOTATION VALIDATION — project summary")
|
||||
add(RULE_THICK)
|
||||
add("")
|
||||
add(string.format("Atoms: %d", total_atoms))
|
||||
add(string.format("Annotations: %d", total_annots))
|
||||
add(string.format("Macros: %d", total_macros))
|
||||
add(string.format("Binds: %d", total_binds))
|
||||
add("")
|
||||
add(string.format("Errors: %d", total_errors))
|
||||
add(string.format("Warnings: %d", total_warnings))
|
||||
add("## Module summary"); add("")
|
||||
add("| metric | value |"); add("|--------|-------|")
|
||||
add(string.format("| sources | %d |", #dir_sources))
|
||||
add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
|
||||
#(sa_results.atoms or {}),
|
||||
#(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
|
||||
add(string.format("| annotations | %d |", n_annot))
|
||||
add(string.format("| binds structs | %d |", n_binds))
|
||||
add(string.format("| macro decls | %d |", n_macros))
|
||||
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
|
||||
#(sa_results.findings or {}),
|
||||
#(sa_results.errors or {}),
|
||||
#(sa_results.warnings or {}),
|
||||
#(sa_results.info or {})))
|
||||
add("")
|
||||
|
||||
if total_errors > 0 then
|
||||
add("Per-source error counts:")
|
||||
for _, r in ipairs(all_results) do
|
||||
if #r.errors > 0 then
|
||||
local src_name = source_basename(r.source)
|
||||
add(string.format(" %s : %d error(s)", src_name, #r.errors))
|
||||
-- Sources
|
||||
add("## Sources"); add("")
|
||||
for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
|
||||
add("")
|
||||
|
||||
-- Atoms (annotation)
|
||||
add("## Atoms"); add("")
|
||||
add("| kind | name | source | line |"); add("|------|------|--------|------|")
|
||||
for _, r in ipairs(annot_results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, a in ipairs(r.atoms) do
|
||||
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
|
||||
-- Annotations
|
||||
add("## Annotations"); add("")
|
||||
if #annot_results == 0 then
|
||||
add("_(none)_")
|
||||
else
|
||||
add("| source | line | name | binds | reads | writes |")
|
||||
add("|--------|------|------|-------|-------|--------|")
|
||||
for _, r in ipairs(annot_results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, a in ipairs(r.annots) do
|
||||
local binds = a.binds or "—"
|
||||
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or "—"
|
||||
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or "—"
|
||||
add(string.format("| %s | %d | %s | %s | %s | %s |",
|
||||
src_name, a.line, a.name, binds, reads, writes))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
|
||||
-- Binds_* structs
|
||||
add("## Binds_* structs"); add("")
|
||||
if #annot_results == 0 then
|
||||
add("_(none)_")
|
||||
else
|
||||
for _, r in ipairs(annot_results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, b in ipairs(r.binds) do
|
||||
add(string.format("### %s (%s:%d, %d bytes)",
|
||||
b.name, src_name, b.line, b.bytes))
|
||||
for _, f in ipairs(b.fields) do
|
||||
add(string.format("- `+%d %s`", f.offset, f.name))
|
||||
end
|
||||
add("")
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Macro decls
|
||||
add("## Macro word-count declarations"); add("")
|
||||
if #annot_results == 0 then
|
||||
add("_(none)_")
|
||||
else
|
||||
add("| source | line | macro declaration |")
|
||||
add("|--------|------|-------------------|")
|
||||
for _, r in ipairs(annot_results) do
|
||||
local src_name = source_basename(r.source)
|
||||
for _, m in ipairs(r.macros) do
|
||||
add(string.format("| %s | %d | %s |",
|
||||
src_name, m.line, m.name))
|
||||
end
|
||||
end
|
||||
end
|
||||
add("")
|
||||
|
||||
-- Findings by atom (static-analysis)
|
||||
add("## Static analysis — findings by atom"); add("")
|
||||
local by_atom = {}
|
||||
for _, f in ipairs(sa_results.findings or {}) do
|
||||
by_atom[f.atom] = by_atom[f.atom] or {}
|
||||
by_atom[f.atom][#by_atom[f.atom] + 1] = f
|
||||
end
|
||||
if next(by_atom) == nil then
|
||||
add("_(no findings)_")
|
||||
else
|
||||
for _, a in ipairs(sa_results.atoms or {}) do
|
||||
local fs = by_atom[a.name]
|
||||
if fs then
|
||||
add(string.format("### %s", a.name))
|
||||
for _, f in ipairs(fs) do
|
||||
add(string.format("- `[%s] %s`", f.check, f.msg))
|
||||
end
|
||||
add("")
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Errors / Warnings / Info
|
||||
local function add_findings(label, entries)
|
||||
add(string.format("## %s", label))
|
||||
if #entries == 0 then
|
||||
add("_(none)_")
|
||||
else
|
||||
for _, e in ipairs(entries) do
|
||||
add(string.format("- line %d %s", e.line, e.msg))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
end
|
||||
add_findings("Errors", sa_results.errors or {})
|
||||
add_findings("Warnings", sa_results.warnings or {})
|
||||
add_findings("Info", sa_results.info or {})
|
||||
|
||||
-- Per-atom cycle counts (path-aware)
|
||||
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
|
||||
add("| atom | source | min | max | branches | paths | notes |")
|
||||
add("|------|--------|-----|-----|----------|-------|-------|")
|
||||
local sorted = {}
|
||||
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
|
||||
table.sort(sorted, function(x, y)
|
||||
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
|
||||
end)
|
||||
for _, a in ipairs(sorted) do
|
||||
local p = a.paths or {}
|
||||
local src_name = a.source_path and source_basename(a.source_path) or ""
|
||||
local notes = ""
|
||||
if p.has_loops then notes = notes .. " [loop!]" end
|
||||
if p.unknown_macros and #p.unknown_macros > 0 then
|
||||
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
|
||||
end
|
||||
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
|
||||
a.name, src_name,
|
||||
p.cycles_min or 0, p.cycles_max or 0,
|
||||
p.branches or 0, p.paths or 0, notes))
|
||||
end
|
||||
add("")
|
||||
|
||||
-- Per-source scan summary
|
||||
add("## Per-source scan summary"); add("")
|
||||
for _, src in ipairs(dir_sources) do
|
||||
local src_atoms = {}
|
||||
for _, a in ipairs(sa_results.atoms or {}) do
|
||||
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
|
||||
end
|
||||
if #src_atoms > 0 then
|
||||
local mn, mx = math.huge, -1
|
||||
for _, a in ipairs(src_atoms) do
|
||||
local p = a.paths or {}
|
||||
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
|
||||
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
|
||||
end
|
||||
local path_str
|
||||
if mx > 0 then
|
||||
path_str = string.format(" cycles=%d..%d", mn, mx)
|
||||
else
|
||||
path_str = string.format(" %d cycles", mn)
|
||||
end
|
||||
add(string.format("- `%s` — %d atom%s%s",
|
||||
src.basename, #src_atoms,
|
||||
#src_atoms == 1 and "" or "s", path_str))
|
||||
end
|
||||
end
|
||||
add("")
|
||||
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Orchestration helpers
|
||||
-- REPORT_RENDERERS — data-driven report dispatch (one row per file kind)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- (internal) Re-validate every source in a directory against the canonical corpus projection.
|
||||
--- Calls `annotation.validate()` per source to produce the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
|
||||
--- that the report renderer consumes. Eeach report pass run is reproducible from the corpus.
|
||||
--- Returns the list of module results + the flat list of all results (for the project-wide summary).
|
||||
--- @param ctx PassCtx
|
||||
--- @param dir_sources SourceFile[]
|
||||
--- @return AnnotationResult[], AnnotationResult[]
|
||||
local function lookup_module_results(ctx, dir_sources)
|
||||
local module_results = {}
|
||||
local all_results = {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
if src.scan then
|
||||
local result = annotation.validate(ctx, src, nil)
|
||||
result.source = src.path -- tag for downstream rendering
|
||||
module_results[#module_results + 1] = result
|
||||
all_results[#all_results + 1] = result
|
||||
end
|
||||
end
|
||||
return module_results, all_results
|
||||
end
|
||||
|
||||
--- (internal) Does this module's results contain anything worth emitting?
|
||||
--- @param module_results AnnotationResult[]
|
||||
--- @return boolean
|
||||
local function module_has_content(module_results)
|
||||
for _, r in ipairs(module_results) do
|
||||
if #r.atoms > 0 or #r.annots > 0 or #r.binds > 0
|
||||
or #r.macros > 0 or #r.errors > 0 or #r.warnings > 0 then
|
||||
return true
|
||||
end
|
||||
end
|
||||
return false
|
||||
end
|
||||
|
||||
--- (internal) Log a debug message if `_G[DEBUG_FLAG]` is truthy.
|
||||
--- @param fmt string
|
||||
local function debug_log(fmt, ...)
|
||||
if _G[DEBUG_FLAG] then
|
||||
io.stderr:write(string.format("[%s] " .. fmt, PASS_NAME, ...))
|
||||
end
|
||||
end
|
||||
-- `once = true` means render once at the project level (not per-module).
|
||||
-- `basename(dir_basename)` yields the file's basename for that kind.
|
||||
-- `gather(ctx, dir, dir_sources [, all_modules])` returns the rendered string.
|
||||
local REPORT_RENDERERS = {
|
||||
{
|
||||
name = "atom_meta_report",
|
||||
ext = "md",
|
||||
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
|
||||
once = false,
|
||||
gather = function(ctx, dir, dir_sources)
|
||||
-- Annotations: re-run `annotation.validate()` per source (the existing pattern).
|
||||
local annot_results = {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
if src.scan then
|
||||
local r = annotation.validate(ctx, src, nil)
|
||||
r.source = src.path
|
||||
annot_results[#annot_results + 1] = r
|
||||
end
|
||||
end
|
||||
-- Static-analysis: read stashed projection (no re-validate).
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
|
||||
return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
||||
end,
|
||||
},
|
||||
{
|
||||
name = "atoms",
|
||||
ext = "md",
|
||||
basename = function(dir_basename) return dir_basename .. ".atoms" end,
|
||||
once = false,
|
||||
gather = function(ctx, dir, dir_sources)
|
||||
return render_module_atoms_md(dir, dir_sources,
|
||||
ctx.shared.corpus.word_counts or {})
|
||||
end,
|
||||
},
|
||||
{
|
||||
name = "summary",
|
||||
ext = "md",
|
||||
basename = function(_dir_basename) return "atom_meta_report.summary" end,
|
||||
once = true,
|
||||
gather = function(_ctx, _dir, _dir_sources, all_modules)
|
||||
return render_project_summary(all_modules)
|
||||
end,
|
||||
},
|
||||
}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M — module exports
|
||||
-- M — public pass surface
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
--- Run the report pass.
|
||||
--- Renders one `<dir_basename>.annotations.txt` per source-directory that has content, plus the project-wide `annotation_validation.txt` summary.
|
||||
--- Run the report pass. Emits 1 `atom_meta_report.summary.md` per build + 2 `atom_meta_report.md` + 2 `atoms.md` files per module (duffle + gte_hello).
|
||||
--- Reads `corpus.static_analysis_results` (added in Phase 1) to populate per-module findings without re-running validate().
|
||||
--- @param ctx PassCtx
|
||||
--- @return PassResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
local outputs = {}
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
local by_dir = (corpus and corpus.sources_by_dir) or {}
|
||||
|
||||
-- Module grouping comes from `corpus.sources_by_dir` (the canonical projection).
|
||||
-- Iterate it directly; no private cache, no per-pass stash.
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
local by_dir = (corpus and corpus.sources_by_dir) or {}
|
||||
-- `out_path_root`: when the conventional `out_root` is `build/gen` (any spelling — relative, absolute, separator variants).
|
||||
-- Write the md files to `build/` (parent of `gen/`) instead of nested under `gen/`.
|
||||
-- Mirrors the `gdb_tape_atoms_runtime.gdb` relocation.
|
||||
local function ends_with_gen(p)
|
||||
return type(p) == "string" and (p:match("[/\\]gen[/\\]?$") ~= nil
|
||||
or p == "build/gen" or p == "build\\gen")
|
||||
end
|
||||
local out_root_effective = ends_with_gen(ctx.out_root)
|
||||
and ctx.out_root:gsub("[/\\]gen[/\\]?$", "")
|
||||
or ctx.out_root
|
||||
|
||||
duffle.ensure_dir(ctx.out_root)
|
||||
duffle.ensure_dir(out_root_effective)
|
||||
|
||||
-- Aggregator for the project-wide `once = true` summary renderer.
|
||||
local all_modules = {}
|
||||
|
||||
local all_results_for_summary = {}
|
||||
for dir, dir_sources in pairs(by_dir) do
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
debug_log("dir=%s basename=%s sources=%d\n", dir, dir_basename, #dir_sources)
|
||||
|
||||
if #dir_sources > 0 then
|
||||
local module_results, all_results = lookup_module_results(ctx, dir_sources)
|
||||
for _, r in ipairs(all_results) do
|
||||
all_results_for_summary[#all_results_for_summary + 1] = r
|
||||
-- Per-renderer dispatch for the per-module renderers (once = false).
|
||||
for _, renderer in ipairs(REPORT_RENDERERS) do
|
||||
if not renderer.once then
|
||||
local body = renderer.gather(ctx, dir, dir_sources)
|
||||
local out_path = out_root_effective .. "/" .. renderer.basename(dir_basename) .. "." .. renderer.ext
|
||||
duffle.write_file(out_path, body)
|
||||
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
|
||||
end
|
||||
end
|
||||
|
||||
if module_has_content(module_results) then
|
||||
local out_path = ctx.out_root .. "/" .. dir_basename .. ".annotations.txt"
|
||||
duffle.write_file(out_path, render_module_report(dir, dir_sources, module_results))
|
||||
outputs[#outputs + 1] = { annotations_txt = out_path }
|
||||
else
|
||||
debug_log(" -> no content; skipping\n")
|
||||
-- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
|
||||
local annot_results = {}
|
||||
for _, src in ipairs(dir_sources) do
|
||||
if src.scan then
|
||||
local r = annotation.validate(ctx, src, nil)
|
||||
r.source = src.path
|
||||
annot_results[#annot_results + 1] = r
|
||||
end
|
||||
end
|
||||
local n_annot, n_binds, n_macros = 0, 0, 0
|
||||
for _, r in ipairs(annot_results) do
|
||||
n_annot = n_annot + #r.annots
|
||||
n_binds = n_binds + #r.binds
|
||||
n_macros = n_macros + #r.macros
|
||||
end
|
||||
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
|
||||
all_modules[#all_modules + 1] = {
|
||||
module = dir_basename,
|
||||
atoms = #(sa_results.atoms or {}),
|
||||
annots = n_annot,
|
||||
binds = n_binds,
|
||||
macros = n_macros,
|
||||
findings = #(sa_results.findings or {}),
|
||||
errors = #(sa_results.errors or {}),
|
||||
warnings = #(sa_results.warnings or {}),
|
||||
info = #(sa_results.info or {}),
|
||||
}
|
||||
end
|
||||
|
||||
-- Project-wide renderer (once = true): write the summary file.
|
||||
for _, renderer in ipairs(REPORT_RENDERERS) do
|
||||
if renderer.once then
|
||||
local body = renderer.gather(ctx, nil, nil, all_modules)
|
||||
local out_path = out_root_effective .. "/" .. renderer.basename("") .. "." .. renderer.ext
|
||||
duffle.write_file(out_path, body)
|
||||
outputs[#outputs + 1] = { kind = renderer.name, path = out_path }
|
||||
end
|
||||
end
|
||||
|
||||
if #all_results_for_summary > 0 then
|
||||
local summary_path = ctx.out_root .. "/annotation_validation.txt"
|
||||
duffle.write_file(summary_path, render_project_report(all_results_for_summary))
|
||||
outputs[#outputs + 1] = { summary_txt = summary_path }
|
||||
end
|
||||
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
return { outputs = outputs, errors = {}, warnings = {} }
|
||||
end
|
||||
|
||||
return M
|
||||
|
||||
+118
-275
@@ -39,17 +39,19 @@
|
||||
--- The report header includes `Info: N` alongside Findings / Errors / Warnings, and a dedicated
|
||||
--- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts.
|
||||
---
|
||||
--- The structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`, `control_transfer_delay_slot_use`) skip atoms/components with `debug_skip == true`.
|
||||
--- The `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
|
||||
--- Flagging them as "missing mac_yield" or "BD slot is redundant" is signal noise, not a logic failure.
|
||||
--- Other checks (transfer_hazards, gpu_portstore_shape, abi_handoff, enum_alias_membership, …) still apply to debug_skip declarations because real hazards / typos can still surface in them.
|
||||
---
|
||||
--- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table:
|
||||
--- `["static-analysis"] = {
|
||||
--- module = "passes.static_analysis",
|
||||
--- kind = "diagnostic",
|
||||
--- deps = {"word-counts", "components"},
|
||||
--- out = { { kind = "report", path_template = "<out_root>/<basename>.static_analysis.txt" } }
|
||||
--- }
|
||||
--- `kind = "diagnostic"` keeps every finding visible in the report; the orchestrator does not exit non-zero on static-analysis errors.
|
||||
--- `kind = "diagnostic"` keeps every finding visible in the projection; the orchestrator does not exit non-zero on static-analysis errors.
|
||||
--- Annotation and header-output validation remain build-stopping.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible.
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
@@ -148,6 +150,38 @@ local OUTPUT_EXTENSION = ".static_analysis.txt"
|
||||
--- @field findings Finding[] -- findings for this atom
|
||||
--- @field total_cycles integer -- sum of token cycle costs
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-word-event helpers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- Pick the source-line field that best represents "where in the user's source file is this word?".
|
||||
--
|
||||
-- `word_events` (populated by `passes/emission_model.lua::stamp_root_provenance`) carry four line fields:
|
||||
-- * `call_line` — physical line in the ROOT atom's source (the line of the `mac_X(...)` call site that triggered this emission, or `body_line` for direct words in the atom body)
|
||||
-- * `body_line` — physical line in the body containing the emitted word (the atom body for direct words; the component body for words expanded inside `mac_X(...)`)
|
||||
-- * `def_line` — line of the COMPONENT's declaration in its source file (only meaningful for words emitted inside a component expansion)
|
||||
-- * `line` — body-relative line in the source text (not a physical source line; rarely useful in rendered findings)
|
||||
--
|
||||
-- For component-expanded words (e.g. the BD-slot nop of `jump_reg(R_AtomJmp)` inside `mac_yield()`),
|
||||
-- `body_line` points into the COMPONENT's source file (e.g. `lottes_tape.h:110` for `ac_yield`'s body).
|
||||
-- The user editing their atom body expects the line to point at THEIR source — i.e. the line where `mac_yield()`
|
||||
-- was called (e.g. `hello_gte_tape.c:35`). That line is `call_line`.
|
||||
--
|
||||
-- For direct words in the atom body (no invocation wrapping them), `call_line == body_line` already,
|
||||
-- so `call_line` works for both cases.
|
||||
local function line_for_word_event(ev)
|
||||
if ev == nil then return 0 end
|
||||
return ev.call_line or ev.body_line or ev.line or ev.def_line or 0
|
||||
end
|
||||
|
||||
-- True iff the given atom/component declaration has the bare `atom_dbg_skip` marker.
|
||||
-- Used by the structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`,
|
||||
-- `control_transfer_delay_slot_use`) to exempt runtime-helper declarations (`tape_exit`, `ac_yield`,
|
||||
-- and the `ac_*` macro components) from findings whose contract they intentionally don't satisfy.
|
||||
local function is_runtime_helper(atom)
|
||||
return atom and atom.debug_skip == true
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- classify_tokens — per-token classification
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -551,11 +585,12 @@ local function append_cu2_finding(atom, event, forward, transition,
|
||||
local event_ident = event.encoder or event.ident or "?"
|
||||
local policy = duffle.CU2_TRANSITION_POLICY or {}
|
||||
local evidence = policy.evidence or {}
|
||||
local event_line = line_for_word_event(event)
|
||||
atom.paths.hazards[#atom.paths.hazards + 1] = {
|
||||
check = "transfer_hazards",
|
||||
kind = kind,
|
||||
atom = atom.name,
|
||||
line = event.body_line or event.line or event.def_line or 0,
|
||||
line = event_line,
|
||||
source = event.def_path or event.source or "",
|
||||
relation_id = "mtc0_cu2_visibility",
|
||||
semantic = "MTC0",
|
||||
@@ -619,10 +654,11 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
|
||||
|
||||
local gap = ev_word - transition.producer_word - 1
|
||||
local target = transition.target_state
|
||||
local event_line = line_for_word_event(event)
|
||||
if target == "unknown" then
|
||||
append_cu2_finding(atom, event, forward, transition, gap, "info", "unknown",
|
||||
string.format("%s at line %d uses COP2 after an MTC0 Status write whose CU2 value is unknown (gap=%d, configured boundary=%d)"
|
||||
, atom.name, event.body_line or event.line or event.def_line or 0
|
||||
, atom.name, event_line
|
||||
, gap, transition.required
|
||||
)
|
||||
)
|
||||
@@ -635,7 +671,7 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
|
||||
local verb = target == "enabled" and "enable" or "disable"
|
||||
append_cu2_finding(atom, event, forward, transition, gap, "warning", "conservative",
|
||||
string.format("%s at line %d uses COP2 before the SR.CU2 %s transition has settled (gap=%d, required=%d; timing is conservative)"
|
||||
, atom.name, event.body_line or event.line or event.def_line or 0
|
||||
, atom.name, event_line
|
||||
, verb, gap, transition.required
|
||||
)
|
||||
)
|
||||
@@ -653,7 +689,7 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
|
||||
string.format(
|
||||
"%s at line %d: COP2 unavailable after SR.CU2 was disabled"
|
||||
.. " (gap=%d, required=%d)",
|
||||
atom.name, event.body_line or event.line or event.def_line or 0,
|
||||
atom.name, event_line,
|
||||
gap, transition.required))
|
||||
forward.cu2_state = "disabled"
|
||||
end
|
||||
@@ -708,7 +744,7 @@ local function analyze_hardware_relations(atom)
|
||||
|
||||
for _, ev in ipairs(events) do
|
||||
local ev_ident = ev.encoder or ev.ident or "?"
|
||||
local ev_line = ev.body_line or ev.line or ev.def_line or 0
|
||||
local ev_line = line_for_word_event(ev)
|
||||
local ev_source = ev.def_path or ev.source or ""
|
||||
local ev_args = ev.args or {}
|
||||
-- `word_events` use `i` as the 0-based word index across the entire expansion.
|
||||
@@ -867,7 +903,7 @@ local function analyze_hardware_relations(atom)
|
||||
|
||||
-- ── 4. Update semantic role state and stage post-command latch relations. ──
|
||||
-- A GTE command emits outputs with semantic roles (latest_screen_xy, otz, latest_color, etc.) per `duffle.GTE_COMMAND_OUTPUTS`.
|
||||
-- The walker records these on `forward_state.post_command_roles[<register>]` so the `gte_result_position` reader can later detect a reader that picks the wrong register.
|
||||
-- The walker records these on `forward_state.post_command_roles[<register>]` so the `gte_role_mismatch` reader can later detect a reader that picks the wrong register.
|
||||
--
|
||||
-- The walker also stages POST-COMMAND LATCH relations (kind = "command_latch_input"): a subsequent MTC2/CTC2 overwrite of a latched output before the measured boundary is a hazard.
|
||||
-- The relation kind is intentionally separate from the preceding MTC2 → command relation (`MTC2` / `CTC2` / `LWC2`).
|
||||
@@ -980,7 +1016,7 @@ local function check_gte_input_latch(atom, _pipe_ctx, findings)
|
||||
end
|
||||
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
-- Check #1e: gte_result_position (READER for forward_state semantic roles).
|
||||
-- Check #1e: gte_role_mismatch (READER for forward_state semantic roles).
|
||||
--
|
||||
-- A GTE command emits outputs with semantic roles (latest_screen_xy, otz, latest_color, etc.) per `duffle.GTE_COMMAND_OUTPUTS`.
|
||||
-- The forward walker records `forward_state.post_command_roles[<register>]` after each command.
|
||||
@@ -988,43 +1024,16 @@ end
|
||||
-- A subsequent MFC2 (or any encoder that reads a C2 register) that picks the WRONG register for the active role emits a `result_role_mismatch` warning.
|
||||
-- For example, reading `C2_SXY0` after RTPS is wrong: the `latest_screen_xy` role is `C2_SXY2`.
|
||||
--
|
||||
-- Note: the OLD `gte_result_position` check also emitted table-gap info findings for `_post_<cmd>` components missing a row in `duffle.GTE_COMPONENT_RESULT_CONTRACTS`. That table-gap check was based on the `_post_<cmd>` NAMING convention rather than hardware truth, and was removed (the user did not want naming to encode ordering semantics; a proper `atom_info` directive for ordering semantics is a future TODO).
|
||||
--
|
||||
-- The first `transfer_hazards` reader comment above records the projection contract.
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
local function check_gte_result_position(atom, _pipe_ctx, findings)
|
||||
local function check_gte_role_mismatch(atom, _pipe_ctx, findings)
|
||||
local forward = atom.paths and atom.paths.forward_state
|
||||
if not forward or not forward.post_command_roles then return end
|
||||
local events = atom.paths.word_events or {}
|
||||
|
||||
-- Build a set of known _post_<cmd> component names whose contract rows we have to verify
|
||||
-- (table-gap detection: a missing row key is itself an info finding).
|
||||
-- The names are the BODY-LEVEL component calls that appear in atom body text;
|
||||
-- The walker doesn't expose body tokens to the reader, so we scan the events' root_call_text.
|
||||
local contracts = duffle.GTE_COMPONENT_RESULT_CONTRACTS or {}
|
||||
local component_names_seen = {}
|
||||
for _, ev in ipairs(events) do
|
||||
local root_call = ev.root_call_text or ev.call_text or ""
|
||||
local name = root_call:match("^([%w_]+)") or ""
|
||||
if name:find("_post_") then component_names_seen[name] = true end
|
||||
end
|
||||
for component_name in pairs(component_names_seen) do
|
||||
-- Strip any trailing parenthesized argument list / whitespace.
|
||||
local bare = component_name:match("^([%w_]+)") or component_name
|
||||
if contracts[bare] == nil then
|
||||
findings[#findings + 1] = {
|
||||
check = "gte_result_position",
|
||||
kind = "info",
|
||||
atom = atom.name,
|
||||
line = 0,
|
||||
source = "",
|
||||
relation_id = "table_gap",
|
||||
component_name = bare,
|
||||
msg = string.format("%s: component %q has no GTE_COMPONENT_RESULT_CONTRACTS row (unknown _post_<cmd> contract)"
|
||||
, atom.name, bare),
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
-- For each word event whose encoder is `gte_mv_from_data_r`, look up the register being read in `forward_state.post_command_roles`.
|
||||
-- If a role is set, the reader's register must match the role's register (the registered "latest_<role>" target).
|
||||
for _, ev in ipairs(events) do
|
||||
@@ -1047,11 +1056,12 @@ local function check_gte_result_position(atom, _pipe_ctx, findings)
|
||||
-- This is a semantic mismatch.
|
||||
if reg ~= latest_screen_xy_entry.command_register
|
||||
and (reg == "C2_SXY0" or reg == "C2_SXY1") then
|
||||
local ev_line = line_for_word_event(ev)
|
||||
findings[#findings + 1] = {
|
||||
check = "gte_result_position",
|
||||
check = "gte_role_mismatch",
|
||||
kind = "warning",
|
||||
atom = atom.name,
|
||||
line = ev.body_line or ev.line or ev.def_line or 0,
|
||||
line = ev_line,
|
||||
source = ev.def_path or ev.source or "",
|
||||
relation_id = "result_role_mismatch",
|
||||
semantic = "result_position",
|
||||
@@ -1062,7 +1072,7 @@ local function check_gte_result_position(atom, _pipe_ctx, findings)
|
||||
producer_word = latest_screen_xy_entry.producer_word,
|
||||
producer_line = latest_screen_xy_entry.producer_line,
|
||||
msg = string.format("%s at line %d: reading %s after %s but the %s role is C2_SXY2 (not %s)"
|
||||
, atom.name, ev.body_line or ev.line or ev.def_line or 0
|
||||
, atom.name, ev_line
|
||||
, reg, latest_screen_xy_entry.command
|
||||
, latest_screen_xy_entry.role
|
||||
, reg),
|
||||
@@ -1084,6 +1094,13 @@ end
|
||||
-- Branch/jump delay-slot NOPs belong to `control_transfer_delay_slot_use`, so this check leaves them unclassified.
|
||||
-- The fixed `mac_yield()` handshake (`jump_reg(R_AtomJmp), nop`) is preserved as suppressed.
|
||||
--
|
||||
-- Both classifications emit at `info` severity: `modeled-required` documents the model boundary and `modeled-redundant`
|
||||
-- is a soft observation ("you have a redundant nop; consider replacing it").
|
||||
-- Neither is a logic failure, so neither rises to `warning`.
|
||||
--
|
||||
-- `atom_dbg_skip` runtime helpers (`tape_exit`, `ac_yield`, the `ac_*` macro components) are exempt:
|
||||
-- their structural nops are part of the fixed handshake and not author choices.
|
||||
--
|
||||
-- The first `transfer_hazards` reader comment above records the projection contract.
|
||||
-- ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -1091,6 +1108,9 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
|
||||
local forward = atom.paths and atom.paths.forward_state
|
||||
local events = atom.paths.word_events or {}
|
||||
if not events or #events == 0 then return end
|
||||
-- Runtime-helper atoms / components (e.g. tape_exit, ac_yield) carry `debug_skip = true` from the bare
|
||||
-- `atom_dbg_skip` marker; their structural nops are part of the fixed handshake and not author choices.
|
||||
if is_runtime_helper(atom) then return end
|
||||
|
||||
-- The walker does not currently snapshot the pending state per event; we replay the same forward walk cheaply here.
|
||||
-- The replay is observation-only (no staging); the only output is one finding per non-BD-slot nop with its classification.
|
||||
@@ -1100,20 +1120,16 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
|
||||
local ev_ident = ev.encoder or ""
|
||||
local ev_args = ev.args or {}
|
||||
local ev_word = ev.i or 0
|
||||
local ev_line = line_for_word_event(ev)
|
||||
|
||||
-- Classify the nop BEFORE its event is applied to the pending state.
|
||||
if ev_ident == "nop" and prev_ev ~= nil then
|
||||
-- Skip BD-slot nops: they are exclusively owned by control_transfer_delay_slot_use.
|
||||
local prev_ident = prev_ev.encoder or ""
|
||||
local prev_args = prev_ev.args or {}
|
||||
local bd_policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {}
|
||||
local is_bd_slot = false
|
||||
local policy = bd_policies[prev_ident]
|
||||
if policy then
|
||||
local arg1 = prev_args[1]
|
||||
local suppressed = policy.suppress_arg1 and policy.suppress_arg1[arg1] or nil
|
||||
if not suppressed then is_bd_slot = true end
|
||||
end
|
||||
-- Every BD-slot nop is structural; this check never reports on it.
|
||||
-- (The earlier `if not suppressed then is_bd_slot = true end` form inverted the suppression — the `mac_yield()` handshake's `jump_reg(R_AtomJmp)` was incorrectly flagged.)
|
||||
local prev_ident = prev_ev.encoder or ""
|
||||
local bd_policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {}
|
||||
local is_bd_slot = bd_policies[prev_ident] ~= nil
|
||||
if not is_bd_slot then
|
||||
-- Find a pending modeled relation that this nop would retire.
|
||||
local retired = nil
|
||||
@@ -1157,7 +1173,7 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
|
||||
check = "hazard_nop_use",
|
||||
kind = "info",
|
||||
atom = atom.name,
|
||||
line = ev.body_line or ev.line or ev.def_line or 0,
|
||||
line = ev_line,
|
||||
source = ev.def_path or ev.source or "",
|
||||
nop_classification = "modeled-required",
|
||||
nop_word_index = ev_word,
|
||||
@@ -1165,7 +1181,7 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
|
||||
producer_destination = retired.destination,
|
||||
consumer_token = would_be_consumer or "<would-be-consumer>",
|
||||
msg = string.format("%s at line %d: nop at word %d is modeled-required (retires %s for %s)"
|
||||
, atom.name, ev.body_line or ev.line or ev.def_line or 0, ev_word, retired.relation.id, retired.destination
|
||||
, atom.name, ev_line, ev_word, retired.relation.id, retired.destination
|
||||
),
|
||||
}
|
||||
else
|
||||
@@ -1173,16 +1189,16 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
|
||||
local slot_kind = "plain"
|
||||
findings[#findings + 1] = {
|
||||
check = "hazard_nop_use",
|
||||
kind = "warning",
|
||||
kind = "info",
|
||||
atom = atom.name,
|
||||
line = ev.body_line or ev.line or ev.def_line or 0,
|
||||
line = ev_line,
|
||||
source = ev.def_path or ev.source or "",
|
||||
nop_classification = "modeled-redundant",
|
||||
nop_word_index = ev_word,
|
||||
retired_relation = nil,
|
||||
slot_kind = slot_kind,
|
||||
msg = string.format("%s at line %d: nop at word %d is modeled-redundant (no pending modeled relation)"
|
||||
, atom.name, ev.body_line or ev.line or ev.def_line or 0, ev_word
|
||||
, atom.name, ev_line, ev_word
|
||||
),
|
||||
}
|
||||
end
|
||||
@@ -1254,6 +1270,9 @@ end
|
||||
-- Suppress the finding when `policy.suppress_arg1[first_arg]` is non-nil.
|
||||
-- The only current suppression is `jump_reg(R_AtomJmp)`, the fixed `mac_yield()` handshake.
|
||||
--
|
||||
-- `atom_dbg_skip` runtime helpers (`tape_exit`, `ac_yield`, the `ac_*` macro components) are exempt:
|
||||
-- their BD slots are part of the fixed handshake (`jump_reg(rret_addr), nop` for tape_exit, `jump_reg(R_AtomJmp), nop` for ac_yield).
|
||||
--
|
||||
-- `pipe_ctx` is unused; the uniform `(atom, pipe_ctx, findings)` signature is preserved so the check plugs into
|
||||
-- the existing CHECK_RULES dispatch without modifying the per-atom loop or analyze_atom_paths.
|
||||
-- `passes/emission_model` already normalizes `nop2` to two `nop` events and `atom_label` to zero events, so no special-case branching is needed for either.
|
||||
@@ -1262,6 +1281,9 @@ end
|
||||
local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings)
|
||||
local events = atom.paths.word_events or {}
|
||||
if not events or #events == 0 then return end
|
||||
-- Runtime-helper atoms / components (e.g. tape_exit, ac_yield) carry `debug_skip = true` from the bare
|
||||
-- `atom_dbg_skip` marker; their structural BD slots are part of the fixed handshake.
|
||||
if is_runtime_helper(atom) then return end
|
||||
local policies = duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES or {}
|
||||
for event_idx, event in ipairs(events) do
|
||||
-- Canonical word_events use `encoder` as the leading identifier of the emitting token).
|
||||
@@ -1276,9 +1298,9 @@ local function check_control_transfer_delay_slot_use(atom, pipe_ctx, findings)
|
||||
local slot = events[event_idx + 1]
|
||||
local slot_ident = slot and (slot.encoder or slot.ident) or "<missing>"
|
||||
if slot == nil or (slot.encoder or slot.ident) == "nop" then
|
||||
-- Each word event carries `body_line` as the physical source line.
|
||||
-- Use `body_line`, then `def_line`, then 0.
|
||||
local ev_line = event.body_line or event.line or event.def_line or 0
|
||||
-- Prefer `call_line` (the line of the `mac_X(...)` call site in the atom body) so the rendered
|
||||
-- finding points at the user's source, not at the vendored component body.
|
||||
local ev_line = line_for_word_event(event)
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = ev_line,
|
||||
@@ -1303,8 +1325,17 @@ end
|
||||
--- Empty bodies are not currently flagged — runtime infrastructure atoms like
|
||||
--- `MipsAtom_(yield) { mac_yield() }` and `MipsAtom_(tape_exit) { jump_reg(rret_addr), nop }`
|
||||
--- are valid as-is; mac_yield at the end is the contract.
|
||||
---
|
||||
--- Runtime helpers carrying the bare `atom_dbg_skip` marker (`tape_exit`, `ac_yield`, the `ac_*` macro components) are exempt:
|
||||
--- they intentionally do not follow the standard "1 yield at the end" contract. `tape_exit` performs its own `jump_reg(rret_addr),
|
||||
--- nop` to return from the tape runner; `ac_yield` IS the `mac_yield()` implementation.
|
||||
--- Flagging them as "missing mac_yield" is signal noise, not a logic failure.
|
||||
---
|
||||
--- Uses the standard `(atom, pipe_ctx, findings)` signature; `pipe_ctx` is unused.
|
||||
local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
||||
-- Runtime-helper atoms / components (e.g. tape_exit, ac_yield) carry `debug_skip = true` from the bare
|
||||
-- `atom_dbg_skip` marker; they intentionally break the "1 yield at the end" contract.
|
||||
if is_runtime_helper(atom) then return end
|
||||
-- Per-kind semantics:
|
||||
-- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job.
|
||||
-- MipsAtomComp_ (bare static-array component): ZERO mac_yield.
|
||||
@@ -1935,7 +1966,7 @@ end
|
||||
local CHECK_RULES = {
|
||||
{ name = "transfer_hazards", per_atom = check_transfer_hazards },
|
||||
{ name = "gte_input_latch", per_atom = check_gte_input_latch },
|
||||
{ name = "gte_result_position", per_atom = check_gte_result_position },
|
||||
{ name = "gte_role_mismatch", per_atom = check_gte_role_mismatch },
|
||||
{ name = "hazard_nop_use", per_atom = check_hazard_nop_use },
|
||||
{ name = "control_transfer_delay_slot_use",per_atom = check_control_transfer_delay_slot_use},
|
||||
{ name = "mac_yield_uniformity", per_atom = check_mac_yield_uniformity },
|
||||
@@ -2179,203 +2210,6 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
||||
}
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Per-directory output: build/gen/<dir_basename>.static_analysis.txt
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Per-directory emit. Aggregates atoms + findings across every source in `dir_sources`
|
||||
--- and writes a single report to `<out_root>/<dir_basename>.static_analysis.txt`.
|
||||
--- Called only when at least one atom was found (the caller in M.run handles the skip).
|
||||
---
|
||||
--- `info` is finding-level info only (kind == "info" findings); the scanned/cycles summary rows
|
||||
--- live in `summaries` and are rendered as trailing summary lines after `Module findings:`.
|
||||
local function emit_module_static_analysis_txt(ctx, dir, dir_sources, atoms, findings, errors, warnings, info, summaries)
|
||||
-- Module basename = last component of `dir` ("code/duffle" -> "duffle").
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
local out_path = ctx.out_root .. "/" .. dir_basename .. ".static_analysis.txt"
|
||||
duffle.ensure_dir(ctx.out_root)
|
||||
|
||||
local lines = {}
|
||||
local function add(s) lines[#lines + 1] = s end
|
||||
|
||||
add("========================================================")
|
||||
add("STATIC ANALYSIS PASS -- module " .. dir_basename)
|
||||
add("========================================================")
|
||||
add(string.format("Sources: %d", #dir_sources))
|
||||
for _, s in ipairs(dir_sources) do
|
||||
add(" " .. s.path)
|
||||
end
|
||||
add("")
|
||||
|
||||
-- Tally atoms by kind for the header summary
|
||||
local n_atoms, n_bare, n_proc = 0, 0, 0
|
||||
for _, a in ipairs(atoms) do
|
||||
n_atoms = n_atoms + 1
|
||||
if a.kind == "comp_bare" then n_bare = n_bare + 1
|
||||
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
|
||||
end
|
||||
end
|
||||
local header_atoms = string.format("Atoms: %d", n_atoms)
|
||||
if n_bare > 0 or n_proc > 0 then
|
||||
header_atoms = header_atoms .. string.format(" (atoms: %d, comp_bare: %d, comp_proc: %d)",
|
||||
n_atoms - n_bare - n_proc, n_bare, n_proc)
|
||||
end
|
||||
-- Header carries the per-severity counts; info is its own column, not a warning.
|
||||
-- (`Info: N` is the byte-asserted field that the focused test matches; do not collapse it into Warnings.)
|
||||
add(string.format("%s Findings: %d Errors: %d Warnings: %d Info: %d",
|
||||
header_atoms, #findings, #errors, #warnings, #info))
|
||||
add("")
|
||||
|
||||
-- Group findings by atom (with source prefix when multi-source module)
|
||||
local multi_source = #dir_sources > 1
|
||||
local by_atom = {}
|
||||
for _, f in ipairs(findings) do
|
||||
by_atom[f.atom] = by_atom[f.atom] or {}
|
||||
by_atom[f.atom][#by_atom[f.atom] + 1] = f
|
||||
end
|
||||
|
||||
if next(by_atom) == nil then
|
||||
add(" (no findings -- every atom passed all checks)")
|
||||
else
|
||||
add("── Findings by atom ─────────────────────────────────────")
|
||||
for _, a in ipairs(atoms) do
|
||||
local fs = by_atom[a.name]
|
||||
if fs then
|
||||
local label = a.name
|
||||
if multi_source and a.source_path then
|
||||
label = string.format("%s (%s)", a.name, a.source_path:match("([^/\\]+)$") or a.source_path)
|
||||
end
|
||||
add(string.format(" %s line %d", label, a.line))
|
||||
for _, f in ipairs(fs) do
|
||||
add(string.format(" [%s] %s", f.check, f.msg))
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
add("")
|
||||
add("── Errors ──────────────────────────────────────────────")
|
||||
if #errors == 0 then add(" (none)") end
|
||||
for _, e in ipairs(errors) do
|
||||
add(string.format(" X line %d %s", e.line, e.msg))
|
||||
end
|
||||
|
||||
add("")
|
||||
add("── Warnings ────────────────────────────────────────────")
|
||||
if #warnings == 0 then add(" (none)") end
|
||||
for _, w in ipairs(warnings) do
|
||||
add(string.format(" ! line %d %s", w.line, w.msg))
|
||||
end
|
||||
|
||||
-- Finding-level Info section.
|
||||
-- Rendered between Warnings and the per-atom cycle table so the next `── ` line after `── Info` is the per-atom cycle counts section;
|
||||
-- the trailing scan/cycle summary rows (rendered after Module findings) stay outside this section.
|
||||
add("")
|
||||
add("── Info ────────────────────────────────────────────────")
|
||||
if #info == 0 then add(" (none)") end
|
||||
for _, i_ in ipairs(info) do
|
||||
add(string.format(" i line %d %s", i_.line, i_.msg))
|
||||
end
|
||||
|
||||
-- Per-atom cycle counts (path-aware). For each atom:
|
||||
-- min = shortest path through the body (earliest exit)
|
||||
-- max = longest path through the body (full fall-through)
|
||||
-- br = number of branch instructions
|
||||
-- paths = number of distinct paths reached
|
||||
-- Both min and max are best-case (no stalls); BD-slot nops are absorbed into branch costs (MIPS semantics).
|
||||
add("")
|
||||
add("── Per-atom cycle counts (path-aware, best case, no stalls) ─")
|
||||
if #atoms == 0 then
|
||||
add(" (no atoms)")
|
||||
else
|
||||
-- Sort atoms by max cycles descending for quick scanning.
|
||||
local sorted = {}
|
||||
for _, a in ipairs(atoms) do sorted[#sorted + 1] = a end
|
||||
table.sort(sorted, function(x, y) return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0) end)
|
||||
for _, a in ipairs(sorted) do
|
||||
local p = a.paths or {}
|
||||
local br_count = p.branches or 0
|
||||
local path_count = p.paths or 0
|
||||
local loops_tag = p.has_loops and " [loop!]" or ""
|
||||
local unknown_tag = ""
|
||||
if p.unknown_macros and #p.unknown_macros > 0 then
|
||||
unknown_tag = string.format(" [unknown: %s]",
|
||||
table.concat(p.unknown_macros, ", "))
|
||||
end
|
||||
local name_label = a.name
|
||||
if multi_source and a.source_path then
|
||||
name_label = string.format("%s (%s)", a.name, a.source_path:match("([^/\\]+)$") or a.source_path)
|
||||
end
|
||||
if br_count > 0 then
|
||||
add(string.format(" %-44s min=%4d max=%4d br=%d paths=%d (line %d)%s%s",
|
||||
name_label, p.cycles_min or 0, p.cycles_max or 0, br_count, path_count,
|
||||
a.line, loops_tag, unknown_tag))
|
||||
else
|
||||
add(string.format(" %-44s %4d cycles (line %d, no branches)%s%s",
|
||||
name_label, p.cycles_min or 0, a.line, loops_tag, unknown_tag))
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
add("")
|
||||
add("── Per-source scan summary ──────────────────────────────")
|
||||
-- One line per source that contributed atoms.
|
||||
-- The line includes the source basename + per-source atom count + (if path-aware cycle data is present) the min..max cycle range.
|
||||
-- Sources with 0 atoms are skipped (they're just header files that declared no MipsAtom_ — they're already listed in the module's "Sources:" section above).
|
||||
for _, src in ipairs(dir_sources) do
|
||||
local src_atoms = {}
|
||||
for _, a in ipairs(atoms) do
|
||||
if a.source_path == src.path then
|
||||
src_atoms[#src_atoms + 1] = a
|
||||
end
|
||||
end
|
||||
if #src_atoms == 0 then
|
||||
goto continue
|
||||
end
|
||||
local atom_count = #src_atoms
|
||||
local mn, mx = math.huge, -1
|
||||
for _, a in ipairs(src_atoms) do
|
||||
local p = a.paths or {}
|
||||
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
|
||||
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
|
||||
end
|
||||
local path_str
|
||||
if mx > 0 then
|
||||
path_str = string.format(" cycles=%d..%d", mn, mx)
|
||||
else
|
||||
path_str = string.format(" %d cycles", mn)
|
||||
end
|
||||
add(string.format(" %-30s %d atom%s%s",
|
||||
src.basename, atom_count,
|
||||
atom_count == 1 and "" or "s",
|
||||
path_str))
|
||||
::continue::
|
||||
end
|
||||
|
||||
-- Module-level findings summary (across all sources).
|
||||
-- Info has its own count; it remains separate from warnings.
|
||||
local total_errs = #errors
|
||||
local total_warns = #warnings
|
||||
local total_infos = #info
|
||||
add("")
|
||||
add(string.format("Module findings: %d error(s), %d warning(s), %d info", total_errs, total_warns, total_infos))
|
||||
|
||||
-- Per-source "scanned:" / "cycles:" summary lines (each line includes the source basename for traceability).
|
||||
-- These are kept SEPARATE from the finding-level Info section above so the report's Info section is signal-only
|
||||
-- (true findings), not a mix of findings + rollups.
|
||||
-- The downstream test (`test_control_transfer_delay_slot.lua`)
|
||||
-- asserts that the Info section contains NEITHER `scanned:` NOR `cycles:` lines.
|
||||
if summaries and #summaries > 0 then
|
||||
add("")
|
||||
for _, s in ipairs(summaries) do
|
||||
add(string.format(" %s", s.msg))
|
||||
end
|
||||
end
|
||||
|
||||
duffle.write_file(out_path, table.concat(lines, "\n") .. "\n")
|
||||
return out_path
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- M.run — orchestrator entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -2431,21 +2265,30 @@ function M.run(ctx)
|
||||
for _, s in ipairs(result.summaries or {}) do dir_summaries[#dir_summaries + 1] = s end
|
||||
end
|
||||
|
||||
-- Skip directories with zero atoms. A directory with only headers / no MipsAtom_ is "nothing to report".
|
||||
if #all_atoms == 0 then
|
||||
-- Still aggregate errors/warnings/info so orchestrator sees them, but don't write a file.
|
||||
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end
|
||||
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end
|
||||
for _, i_ in ipairs(dir_info) do info[#info + 1] = i_ end
|
||||
else
|
||||
local out_path = emit_module_static_analysis_txt(ctx, dir, dir_sources, all_atoms, all_findings, dir_errors, dir_warnings, dir_info, dir_summaries)
|
||||
if out_path then
|
||||
table.insert(outputs, { static_analysis_txt = out_path })
|
||||
end
|
||||
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end
|
||||
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end
|
||||
for _, i_ in ipairs(dir_info) do info[#info + 1] = i_ end
|
||||
end
|
||||
-- Stash per-module results on the corpus for `report.lua` to consume.
|
||||
-- Avoids re-running validate() in the report pass + avoids rebuilding corpus_pipe_ctx.
|
||||
-- Pattern matches `corpus.atoms_by_name` / `corpus.word_counts` / `corpus.components`
|
||||
-- (one writer: `static_analysis.lua`; one reader: `report.lua`).
|
||||
-- Module basename = last component of `dir` ("code/duffle" -> "duffle").
|
||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
||||
corpus.static_analysis_results = corpus.static_analysis_results or {}
|
||||
corpus.static_analysis_results[dir_basename] = {
|
||||
atoms = all_atoms,
|
||||
findings = all_findings,
|
||||
errors = dir_errors,
|
||||
warnings = dir_warnings,
|
||||
info = dir_info,
|
||||
summaries = dir_summaries,
|
||||
sources = dir_sources,
|
||||
}
|
||||
|
||||
-- Aggregate per-dir errors/warnings/info into the orchestrator totals.
|
||||
-- Hoisted out of any per-dir file-emit so `report.lua` can drop the on-disk file emitter without losing the cross-module rollup.
|
||||
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end
|
||||
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end
|
||||
for _, i_ in ipairs(dir_info) do info [#info + 1] = i_ end
|
||||
-- (No per-dir emit: per-module findings are stashed on `corpus.static_analysis_results` above.
|
||||
-- `report.lua` reads that projection to render `<module>.atom_meta_report.md` without re-running validate().)
|
||||
end
|
||||
|
||||
-- Result exposes at least {outputs, errors, warnings, info}.
|
||||
|
||||
@@ -2,17 +2,14 @@
|
||||
---
|
||||
--- Dispatches to pass modules under `scripts/passes/`, resolving dependencies topologically (Kahn's algorithm + cycle detection).
|
||||
---
|
||||
--- **Architecture**:
|
||||
--- - **PASSES table** — declarative dep graph (data, not code).
|
||||
--- - **FLAG_HANDLERS table** — maps CLI flags to handlers.
|
||||
--- - **parse_args** → **build_ctx** (resolves unity/direct includes or exact sources; no semantic scanning) → **topo_sort** → **dispatch_passes**.
|
||||
--- Architecture:
|
||||
--- - PASSES table: Declarative dep graph (data, not code).
|
||||
--- - FLAG_HANDLERS table: Maps CLI flags to handlers.
|
||||
--- - parse_args → build_ctx (resolves unity/direct includes or exact sources) → topo_sort → dispatch_passes.
|
||||
--- - The first pass in the dep graph is `scan-source` (see `passes/scan_source.lua`).
|
||||
--- It calls `duffle.scan_source` once per source to produce the fat `SourceScan` payload, which is attached to each `src.scan`.
|
||||
--- Every other pass that reads source structure depends on `scan-source` and consumes `src.scan` as a read-only.
|
||||
---
|
||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
||||
--- Lua 5.3 compatible.
|
||||
---
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Module-scope requires + package.path setup
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -160,7 +157,7 @@ local PASSES = {
|
||||
report = {
|
||||
module = "passes.report",
|
||||
kind = "report",
|
||||
deps = {"annotation", "static-analysis"},
|
||||
deps = {"annotation", "static-analysis", "atoms-source-map"}, -- +atoms-source-map (consolidated-report-files refactor, 2026-07-26)
|
||||
groups = { "pre-link" },
|
||||
},
|
||||
}
|
||||
@@ -206,7 +203,8 @@ end
|
||||
|
||||
-- Pass-kind taxonomy: Which kinds stop the build on errors?
|
||||
--
|
||||
-- Report severity is independent from process exit policy. A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
|
||||
-- Report severity is independent from process exit policy.
|
||||
-- A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
|
||||
-- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero.
|
||||
-- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true".
|
||||
local PASS_KIND_STOP_ON_ERROR = {
|
||||
|
||||
Reference in New Issue
Block a user