32 Commits
Author SHA1 Message Date
ed 67a84d34f3 oops: endregion 2026-08-14 13:41:45 -04:00
ed baaff12f33 Ideating on "RegUse_" patterned structs for describe register allocatins to mips atom proc. 2026-08-14 12:38:00 -04:00
ed b695056b9a finished reviewing normalize_v3s4 for now 2026-08-14 03:45:34 -04:00
ed 3a4d6304dd static analysis: immeidate field awarenss 2026-08-14 01:22:54 -04:00
ed a535d381ed remove encoding masks from gp (unnecessary, hides errors) 2026-08-14 01:22:36 -04:00
ed c447bfa877 fixes to the reg file allocator, exploring... 2026-08-14 00:43:19 -04:00
ed d88e0d0487 remove mask from mips and gte instruction encoders. missing math changes. 2026-08-13 23:39:35 -04:00
ed 9a6eca6047 more review, made a register file allocator (drafted, kinda iffy, want todo comp-time as well). 2026-08-13 23:39:03 -04:00
ed 5c9c61720f Redesign: Not making local var in MipsAtom_Proc_ or MipsAtomComp_Proc_ have sym tied to proc name. Adjusted parser as well base do that. 2026-08-13 21:42:06 -04:00
ed b8e31123e4 editing/reading. 2026-08-13 21:22:29 -04:00
ed ea3e30a11e oops 2026-08-13 20:51:45 -04:00
ed 37f4712237 gutting nosiy comments. Looking into some atom components.. 2026-08-13 19:55:17 -04:00
ed b699b47b28 intiial review on: resolve_look_at__input_and_sub_proc 2026-08-13 19:42:43 -04:00
ed 640dab7e61 wip: starting to review and update lua metaprogram with more modeling of gte. 2026-08-13 18:51:09 -04:00
ed 4688566767 FINALLY? 2026-08-13 17:42:55 -04:00
ed 5ebaa6e083 still failing 2026-08-13 13:18:27 -04:00
ed 7f0bdefbcb checkpoint nothing 2026-08-13 02:13:47 -04:00
ed d5f28b83ea minor 2026-08-12 22:41:52 -04:00
ed 3ea3e8d105 sssiiighhhh 2026-08-12 20:36:11 -04:00
ed 6b60cef2e8 sigh 2026-08-12 20:30:26 -04:00
ed 77f19321cd pain 2026-08-12 20:24:13 -04:00
ed 2e07665920 Run-Time Library Overview manual 2026-08-12 20:24:05 -04:00
ed 9501bbbcc2 WIP 2026-08-12 20:17:53 -04:00
ed 9b6b5535f5 wip 2026-08-12 20:09:57 -04:00
ed 7807047dc0 Atoms 2-3 work for resolve look at. Don't need OA_ macro so going to stop using. 2026-08-11 21:35:59 -04:00
ed 7daeec0ee3 checkpoint: atom 0-1 works for resolve look at. 2026-08-11 14:05:23 -04:00
ed 3f3b691ac0 Making a proper distinction between atom arenas and atom builders. 2026-08-11 11:25:54 -04:00
ed a2d79d65eb amazing bug 2026-08-11 01:25:40 -04:00
ed bebcc6a585 wip: going to incremnetally test this. 2026-08-11 01:25:09 -04:00
ed ece21ed368 mark current crashing path. 2026-08-10 23:29:38 -04:00
ed 144c605ad8 some more review. not working still. 2026-08-10 23:04:43 -04:00
ed 4afd1af0fd started to review this... 2026-08-10 19:53:34 -04:00
36 changed files with 12732 additions and 1709 deletions
+2 -2
View File
@@ -70,8 +70,8 @@
/* ----------------------------------------------------------------------------
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
*
* The bare `atom_reg` token adjacent to an enum entry in mips.h / lottes_tape.h flags that alias as debug-visible for scan_source's register_alias_registry.
* The C preprocessor strips it to a comment so no runtime symbol is created; the Lua scanner reads the bare token.
* Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
* Lua scanner reads the bare token.
* ----------------------------------------------------------------------------*/
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
+17 -15
View File
@@ -3,7 +3,7 @@
# include "assert.h"
#endif
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
#define static_assert _Static_assert
#define typeof __typeof__
#define typeof_ptr(ptr) typeof((ptr)[0])
@@ -91,12 +91,13 @@
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
#define TSet_(type) type; typedef PtrSet_(type)
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
#define array_decl(type, ...) (type[]){__VA_ARGS__}
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
#define Array_sym(type,len) A ## len ## _ ## type
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
#define Array_(type,len) Array_expand(type,len)
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
#define Bitmask_(b) (1u << b)
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
#define Proc_(symbol) symbol
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
@@ -139,16 +140,15 @@ enum { false = 0, true = 1, true_overflow, };
typedef void Proc_(VoidFn) (void);
#define kilo(n) (C_(U4, n) << 10)
#define mega(n) (C_(U4, n) << 20)
#define giga(n) (C_(U4, n) << 30)
#define tera(n) (C_(U4, n) << 40)
#define Kilo_(n) (C_(U4, n) << 10)
#define Mega_(n) (C_(U4, n) << 20)
#define Giga_(n) (C_(U4, n) << 30)
#define Tera_(n) (C_(U4, n) << 40)
#define null C_(U4, 0)
#define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
#define OT_(field) O_(typeof_ptr(& field), filed))
#define OT_(field) O_(typeof_ptr(& field), field))
#define S_(data) C_(U4, sizeof(data))
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
@@ -185,7 +185,7 @@ def_signed_ops(le, <=)
#define alignas _Alignas
#define alignof _Alignof
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
#define pcast(type, data) (C_(type*, & (data)) [0])
#define C_ptr(type, data) (C_(type*, & (data)) [0])
#define dbg_args(...) __VA_ARGS__
@@ -200,6 +200,8 @@ def_signed_ops(le, <=)
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
#define Jmp_nZero_(cond,label) if (cond) goto label;
#pragma endregion Control Flow & Iteration
#define span_iter(type, iter, m_begin, op, m_end) ( \
@@ -216,16 +218,16 @@ def_signed_ops(le, <=)
typedef Span_(S4);
typedef Span_(U4);
#if 0
#pragma region Debug
#define debug_trap() __builtin_debugtrap()
#define debug_trap() __builtin_trap()
#if BUILD_DEBUG
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
#define assert(cond) if(cond == false){debug_trap();}
#else
#define assert(cond)
# ifndef assert
# include <assert.h>
# endif
#endif
#pragma endregion Debug
#endif
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
+78 -79
View File
@@ -60,8 +60,8 @@ WORD_COUNT(mac_yield_tail, 3)
/* atom_dbg_skip */
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
load_half( rs_x, r_base, O_(V3_S2,x)) \
, load_half( rs_y, r_base, O_(V3_S2,y))
load_half( rs_x, r_base, offset + O_(V3_S2,x)) \
, load_half( rs_y, r_base, offset + O_(V3_S2,y))
WORD_COUNT(mac_load_v2s2, 2)
/* atom_dbg_skip */
@@ -72,9 +72,9 @@ WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
load_word( rs_x, r_base, O_(V3_S4,x)) \
, load_word( rs_y, r_base, O_(V3_S4,y)) \
, load_word( rs_z, r_base, O_(V3_S4,z))
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \
, load_word( rs_z, r_base, offset + O_(V3_S4,z))
WORD_COUNT(mac_load_v3s4, 3)
/* atom_dbg_skip */
@@ -99,6 +99,12 @@ WORD_COUNT(mac_sub_v3s4, 3)
, store_half(rt_height, base, offset + O_(Rect_S2,height))
WORD_COUNT(mac_store_rects2, 4)
/* atom_dbg_skip */
#define mac_load_word_imm(dst, imm) \
load_upper_i(dst, u4_hi(imm)) \
, or_i_self( dst, u4_lo(imm))
WORD_COUNT(mac_load_word_imm, 2)
/* atom_dbg_skip */
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
@@ -149,16 +155,21 @@ WORD_COUNT(mac_gte_store_g4_p3, 1)
/* atom_dbg_skip */
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop \
, gte_cmdw_sqr \
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
WORD_COUNT(mac_gte_sqr_v3, 8)
/* atom_dbg_skip */
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop_slot) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop_slot \
, gte_cmdw_sqr
WORD_COUNT(mac_gte_sqr_v3s4, 5)
/* atom_dbg_skip */
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
gte_mv_to_data_r(r_recip_est, C2_IR0) \
@@ -175,71 +186,58 @@ WORD_COUNT(mac_gte_sqr_v3, 8)
, shift_aright_var(r_dz, r_dz, r_shift)
WORD_COUNT(mac_gte_gpf_scale, 13)
#define mac_normalize_v3s4(...) \
load_word(r_src, R_TapePtr, O_(Binds_NormalizeV3S4,src)) /* pop src ptr (scratch addr) */ \
, load_word(r_dst, R_TapePtr, O_(Binds_NormalizeV3S4,dst)) /* pop dst ptr (scratch addr) */ \
, add_ui_self( R_TapePtr, S_(Binds_NormalizeV3S4)) \
, load_word(r_sx, r_src, O_(V3_S4,x)) \
, load_word(r_sy, r_src, O_(V3_S4,y)) \
, load_word(r_sz, r_src, O_(V3_S4,z)) \
, nop /* load-delay */ /* ── 48-word normalize body (preserved verbatim from ac_normalize_v3s4) ─────── */ /* Stage 1: mtc2 src → IR1/2/3, SQR fires (MAC1/2/3 = IR², IR ← MAC saturated) */ \
, gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop \
, gte_cmdw_sqr /* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS */ \
, gte_mv_from_data_r(r_sq_y, C2_MAC1) /* r_sq_y = MAC1 = sx² */ \
, gte_mv_from_data_r(r_sq_z, C2_MAC2) /* r_sq_z = MAC2 = sy² */ \
, gte_mv_from_data_r(r_recip_est, C2_MAC3) /* r_recip_est = MAC3 = sz² */ \
, nop /* MFC2→GPR load delay (1 slot) */ \
, add_u(r_recip_est, r_recip_est, r_sq_z) /* r_recip_est += sy² */ \
, add_u(r_recip_est, r_recip_est, r_sq_y) /* r_recip_est += sx² (sum = |v|²) */ \
, gte_mv_to_data_r( r_recip_est, C2_LZCS) /* LZCS = |v|² */ \
, nop2 \
, gte_mv_from_data_r(r_lzcr, C2_LZCR) /* r_lzcr = LZCR (count of leading bits) */ \
, nop /* MFC2→GPR load delay (1 slot) */ /* Stage 3: compute shift amount, align |v|² to bit 24, lookup 1/|v| */ \
, and_i( r_lzcr, r_lzcr, -2) /* r_lzcr &= ~1 (force even for halving) */ \
, li_s( r_shift, 31) /* r_shift = 31 */ \
, sub_s( r_shift, r_shift, r_lzcr) /* r_shift = 31 - LZCR */ \
, shift_aright( r_shift, r_shift, 1) /* r_shift = (31 - LZCR) / 2 */ \
, add_si( r_tmp, r_lzcr, -24) /* r_tmp = LZCR - 24 (signed, for branch) */ \
, branch_lt_zero(r_tmp, atom_offset(srav_path, aligned_done)) \
, nop \
, jump_rel( atom_offset(aligned_done, srav_path)) \
, shift_lleft_var(r_recip_est, r_recip_est, r_tmp) /* BD-slot of branch_equal: r_recip_est = |v|² << (LZCR - 24) */ \
, atom_label(srav_path) /* SRAV path: |v|² is small (top bit < bit 24) */ \
, li_s( r_tmp, 24) \
, sub_s( r_tmp, r_tmp, r_lzcr) /* r_tmp = 24 - LZCR */ \
, shift_aright_var(r_recip_est, r_recip_est, r_tmp) /* r_recip_est = |v|² >> (24 - LZCR) */ \
, atom_label(aligned_done) /* Both paths converge here with |v|² aligned to bit 24 */ /* r_recip_est now holds |v|² aligned to bit 24 — convert to byte offset, -64 to skip zero pad. */ \
, add_si( r_recip_est, r_recip_est, -64) \
, shift_lleft( r_recip_est, r_recip_est, 1) /* r_recip_est *= 2 (half-word index) */ /* Reference OUR local sqrtbl via &-address split. Compiler/linker resolves both halves. */ \
, load_upper_i( r_tmp, u4_hi(& gte_normalize_sqr_tbl)) /* lui */ \
, or_i_self( r_tmp, u4_lo(& gte_normalize_sqr_tbl)) /* ori */ \
, add_u( r_tmp, r_tmp, r_recip_est) /* r_tmp = sqrtbl base + byte offset (matches libgte 0x80016118: addu t5,t5,t4) */ \
, load_half( r_recip_est, r_tmp, 0) /* r_recip_est = sqrtbl[r_recip_est] = 1/|v| estimate */ \
, nop /* retire load_half before MTC2 (matches libgte 0x80016120: nop) */ /* Stage 4: mtc2 IR0..3, GPF (MAC = IR0*IR), mfc2 MAC, srav finalize */ \
, gte_mv_to_data_r(r_recip_est, C2_IR0) /* IR0 = 1/|v| estimate */ \
, gte_mv_to_data_r(r_sx, C2_IR1) /* IR1 = src.x */ \
, gte_mv_to_data_r(r_sy, C2_IR2) /* IR2 = src.y */ \
, gte_mv_to_data_r(r_sz, C2_IR3) /* IR3 = src.z */ \
, nop2 /* COP2 transfer latency (2 slots) */ \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_sx, C2_MAC1) /* MAC1 → r_sx (overwrites src.x with raw reciprocal-scaled) */ \
, gte_mv_from_data_r(r_sy, C2_MAC2) \
, gte_mv_from_data_r(r_sz, C2_MAC3) \
, shift_aright_var(r_sx, r_sx, r_shift) \
, shift_aright_var(r_sy, r_sy, r_shift) \
, shift_aright_var(r_sz, r_sz, r_shift) /* ── I/O wrapper tail (~3 words) ───────────────────────────────────────────── */ \
, store_word(r_sx, r_dst, O_(V3_S4,x)) \
, store_word(r_sy, r_dst, O_(V3_S4,y)) \
, store_word(r_sz, r_dst, O_(V3_S4,z)) /* ── atom_reads(R_TapePtr) atom_writes(R_TapePtr) ────────────────────────── */ \
, mac_yield()
WORD_COUNT(mac_normalize_v3s4, 62)
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
load_word(r_t0, r_off, O_(V3_S4,x)) \
, load_word(r_t1, r_off, O_(V3_S4,y)) \
, load_word(r_t2, r_off, O_(V3_S4,z)) \
, store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
, store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
WORD_COUNT(mac_trans_mt3s3s4, 6)
/* atom_dbg_skip */
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
, li_s(r_mag_sq, 31) \
, sub_s(r_mag_sq, r_mag_sq, r_shift) \
, shift_aright(r_mag_sq, r_mag_sq, 1)
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
shift_aright_var(rd_v0, rs_v0, r_shift) \
, shift_aright_var(rd_v1, rs_v1, r_shift) \
, shift_aright_var(rd_v2, rs_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3, 3)
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
shift_aright_var(rds_v0, rds_v0, r_shift) \
, shift_aright_var(rds_v1, rds_v1, r_shift) \
, shift_aright_var(rds_v2, rds_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
gte_mv_to_data_r(to_ir0, C2_IR0) \
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
, gte_mv_to_data_r(to_ir2, C2_IR2) \
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
, LdSlot_ nop_slot1 \
, LdSlot_ nop_slot2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
/* atom_dbg_skip */
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
load_upper_i(reg_transfer, cmd >> 16) \
, or_i_self( reg_transfer, cmd & 0xFFFF) \
mac_load_word_imm(reg_transfer, cmd) \
, store_word( reg_transfer, reg_base, port)
WORD_COUNT(mac_gcmd_push, 3)
@@ -262,6 +260,7 @@ WORD_COUNT(mac_pack_color_word, 3)
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
WORD_COUNT(mac_format_f3_color, 3)
/* atom_dbg_skip */
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
@@ -283,16 +282,16 @@ WORD_COUNT(mac_format_g4_color, 12)
WORD_COUNT(mac_insert_ot_tag, 11)
/* atom_dbg_skip */
#define mac_pad_set_centered_axes(r_state, r_scratch) \
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF) \
, or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF) \
, store_word( r_scratch, r_state, O_(PadState,axes))
#define mac_pad_set_centered_axes(state, scratch) \
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
, or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
, store_word( scratch, state, O_(PadState,axes))
WORD_COUNT(mac_pad_set_centered_axes, 3)
/* atom_dbg_skip */
#define mac_pad_set_id_byte(r_state, r_id, id_value) \
#define mac_pad_set_id_byte(state, r_id, id_value) \
add_ui( r_id, R_0, id_value) \
, store_byte(r_id, r_state, O_(PadState,id))
, store_byte(r_id, state, O_(PadState,id))
WORD_COUNT(mac_pad_set_id_byte, 2)
/* atom_dbg_skip */
+4 -4
View File
@@ -25,14 +25,14 @@
#pragma region duffle
// --- atom: normalize_v3s4 (62 words) ---
// --- atom: normalize_v3s4 (47 words) ---
#define _atom_offset_srav_path_aligned_done 6
#define _atom_offset_aligned_done_srav_path 1
#define _atom_offset_aligned_done_srav_path 3
#define _atom_offset_srav_path_aligned_done 4
enum {
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
};
// --- atom: pad_bios_snapshot (84 words) ---
+12 -12
View File
@@ -8,35 +8,35 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_gcmd_push(MipsAtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_gcmd_push, ab, {
load_upper_i(reg_transfer, cmd >> 16),
or_i_self( reg_transfer, cmd & 0xFFFF),
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_load_word_imm(reg_transfer, cmd),
store_word( reg_transfer, reg_base, port),
})
FI_ Slice_MipsCode ac_store_rgb8(MipsAtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, {
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_byte(rr, base, offset + O_(RGB8,r)),
store_byte(rg, base, offset + O_(RGB8,g)),
store_byte(rb, base, offset + O_(RGB8,b)),
})
FI_ Slice_MipsCode ac_pack_color_word(MipsAtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, {
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(R_AT, (cmd) << 8 | (b)),
or_i_self( R_AT, ((g) << 8) | (r)),
store_word( R_AT, r_base, (off)),
})
FI_ Slice_MipsCode ac_format_f3_color(MipsAtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
FI_ Slice_MipsCode ac_format_g4_color(MipsAtomBuilder_R ab, U4 r_prim_cursor,
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1,
U1 r2, U1 g2, U1 b2,
U1 r3, U1 g3, U1 b3)
MipsAtomComp_Proc_(ac_format_g4_color, ab, {
atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
@@ -44,7 +44,7 @@ MipsAtomComp_Proc_(ac_format_g4_color, ab, {
})
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
I_ Slice_MipsCode ac_insert_ot_tag(MipsAtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, {
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ab, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
+54 -55
View File
@@ -21,7 +21,7 @@
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
* 3. Composite encoders enc_color_word(cmd, r, g, b)
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_width = 8
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
*
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
@@ -74,7 +74,7 @@ enum {
* ============================================================================
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
* NO macro body past this point uses a raw shift or raw mask.
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention from mips.h.
* ============================================================================ */
enum {
gp0_cmd_Nop = 0x00,
@@ -116,21 +116,20 @@ enum {
gp0_cmd_SetDrawOffset = 0xE5,
gp0_cmd_SetMaskBit = 0xE6,
/* bitfield shifts / widths / masks ----
/* bitfield shifts / widths ----
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
gp0_cmd_shift = 24,
gp0_cmd_width = 8,
gp0_cmd_mask = 0xFF,
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
* bits 31..24 = command byte
* bits 23..16 = BLUE
* bits 15..08 = GREEN
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8,
gp0_color_blue_shift = 16, gp0_color_blue_width = 8,
gp0_color_green_shift = 8, gp0_color_green_width = 8,
gp0_color_red_shift = 0, gp0_color_red_width = 8,
};
/* ============================================================================
@@ -143,12 +142,12 @@ enum {
* ============================================================================ */
/* ---- Layer 1.5: per-field encoders ---- */
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
#define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_shift)
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
#define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_shift)
#define enc_gp0_color_r(r) ((r) << gp0_color_red_shift)
#define enc_gp0_color_g(g) ((g) << gp0_color_green_shift)
#define enc_gp0_color_b(b) ((b) << gp0_color_blue_shift)
/* ---- Layer 2: composite encoders ---- */
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
@@ -211,38 +210,38 @@ enum {
gp1_disp_Color24 = 0x1,
gp1_disp_VInterlace = 0x1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2,
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1,
gp1_disp_color_shift = 4, gp1_disp_color_width = 1,
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1,
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12,
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12,
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10,
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10,
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
* (10-bit signed — caller pre-signs and masks with the named mask) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
* (10-bit signed — caller pre-signs) */
gp1_draw_x_shift = 0, gp1_draw_x_width = 10,
gp1_draw_y_shift = 10, gp1_draw_y_width = 10,
};
/* ---- Layer 1.5: GP1 per-field encoders ---- */
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
#define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_shift)
#define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_shift)
#define enc_gp1_disp_color(c) ((c) << gp1_disp_color_shift)
#define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_shift)
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
#define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_shift)
#define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_shift)
#define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_shift)
#define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_shift)
#define enc_gp1_draw_x(x) ((x) << gp1_draw_x_shift)
#define enc_gp1_draw_y(y) ((y) << gp1_draw_y_shift)
/* ---- Layer 2: GP1 composite encoders ---- */
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
@@ -555,14 +554,14 @@ typedef Struct_(Poly_GT4) {
* bits 12..31 = reserved (zero)
* ============================================================================ */
enum {
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
/* ---- Layer 1: TPage bitfield shifts / widths ---- */
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4,
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1,
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2,
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2,
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1,
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1,
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1,
/* TPage color-depth payload values (NOT bit positions — these go in
* the 2-bit field at gp0_tpage_color_depth_shift). */
@@ -581,13 +580,13 @@ enum {
};
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
#define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_shift)
#define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_shift)
#define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_shift)
#define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_shift)
#define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_shift)
#define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_shift)
#define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_shift)
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
@@ -617,17 +616,17 @@ typedef Struct_(TexturePage) { U4 raw; };
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
* ============================================================================ */
enum {
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
/* ---- Layer 1: CLUT bitfield shifts / widths ---- */
gp0_clut_y_shift = 0, gp0_clut_y_width = 6,
gp0_clut_x_shift = 6, gp0_clut_x_width = 9,
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
gp0_clut_cmd_Load4bpp = 0x20,
gp0_clut_cmd_Load8bpp = 0x25,
};
/* ---- Layer 1.5: CLUT per-field encoders ---- */
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
#define enc_gp0_clut_x(x) ((x) << gp0_clut_x_shift)
#define enc_gp0_clut_y(y) ((y) << gp0_clut_y_shift)
/* ---- Layer 2: CLUT composite encoder ---- */
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
+219 -129
View File
@@ -11,7 +11,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
#pragma region MACs (Mips Atom Components)
/* Words: 3; Loads 3 S2 indices from the face array */
FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, {
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
@@ -19,14 +20,14 @@ FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
FI_ Slice_MipsCode ac_gte_store_f3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, {
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
})
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, {
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
@@ -37,7 +38,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, {
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
@@ -47,28 +48,41 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cur
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
* SXY0 still holds v0.screen from the earlier RTPT.
*/
FI_ Slice_MipsCode ac_gte_store_g4_p3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
* Stage 2 of normalize consumes these directly.
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop, gte_cmdw_sqr,
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
gte_mv_from_data_r(r_sq_x, C2_MAC1),
gte_mv_from_data_r(r_sq_y, C2_MAC2),
gte_mv_from_data_r(r_sq_z, C2_MAC3),
})
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ───
* The SQR command always squares IR1/IR2/IR3 — those C2 registers are fixed.
* The GPRs holding the source vector are caller-determined.
* Words: 5 (3 mtc2 + 1 nop hazard + 1 cmd). */
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode nop_slot)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop_slot, gte_cmdw_sqr,
})
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
* Used standalone for "scale vector by scalar".
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, {
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
U4 r_sx, U4 r_sy, U4 r_sz,
U4 r_recip_est, U4 r_shift,
U4 r_dx, U4 r_dy, U4 r_dz)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_recip_est, C2_IR0),
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
@@ -83,6 +97,100 @@ FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r
shift_aright_var(r_dz, r_dz, r_shift),
})
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
* Words: 9. Clobbers: r_t1. */
FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
, U4 r_mtx, U4 r_off
, U4 r_t0, U4 r_t1, U4 r_t2
) MipsAtomComp_Proc_(ab, {
load_word(r_t0, r_off, O_(V3_S4,x)),
load_word(r_t1, r_off, O_(V3_S4,y)),
load_word(r_t2, r_off, O_(V3_S4,z)),
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
})
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32
* per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3
* add). Produces:
* r_shift ← LZCR rounded down to even (clear bit 0)
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
*
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division
* is consistent — no 0.5 loss. The caller branches on LZCR < 24 to decide
* left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
*
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller
* must read it via mfc2 from C2_LZCR; there is no register choice at the
* hardware level. Only the GPR that holds the result is caller-determined. */
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
U4 r_shift,
U4 r_mag_sq,
U4 r_mag_sq_copy
)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
and_i(r_shift, r_shift, gte_lzcr_even_mask),
or_u(r_mag_sq_copy, r_mag_sq, 0),
li_s(r_mag_sq, 31),
sub_s(r_mag_sq, r_mag_sq, r_shift),
shift_aright(r_mag_sq, r_mag_sq, 1),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
, Reg rd_v0, Reg rd_v1, Reg rd_v2
, Reg rs_v0, Reg rs_v1, Reg rs_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rd_v0, rs_v0, r_shift),
shift_aright_var(rd_v1, rs_v1, r_shift),
shift_aright_var(rd_v2, rs_v2, r_shift),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab
, Reg rds_v0, Reg rds_v1, Reg rds_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rds_v0, rds_v0, r_shift),
shift_aright_var(rds_v1, rds_v1, r_shift),
shift_aright_var(rds_v2, rds_v2, r_shift),
})
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
, MipsCode nop_slot1, MipsCode nop_slot2)
MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(to_ir0, C2_IR0),
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
gte_mv_to_data_r(to_ir2, C2_IR2),
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
LdSlot_ nop_slot1,
LdSlot_ nop_slot2,
gte_cmdw_gpf,
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
FI_ Slice_MipsCode gte_mv_from_data_r_mac123(AtomBuilder_R ab
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
)
MipsAtomComp_Proc_(ab, {
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Atom Procs
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
@@ -97,7 +205,8 @@ FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
* Sampling the first value of each octave:
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
@@ -148,132 +257,113 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
};
/* ─── Full normalize (all 4 stages inline) ───
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
*
* Component variants that could apply:
* - `ac_gte_sqr_v3` (line ~56) covers stage 1's `mtc2 IR1/2/3 + nop + gte_cmdw_sqr`.
* We do NOT call it because the inlined version of stage 1 is followed immediately by stage 2's `mfc2 MAC1/2/3` chain
* (the operands of `ac_gte_sqr_v3`'s r_sq_x/r_sq_y/r_sq_z would each require an explicit GPR to receive the MAC result,
* then a move to land in r_recip_est for the partial-sum chain).
* Inlining saves ~3 cycles of `or`-merge + register pressure
* (squared MAC3 lands DIRECTLY in r_recip_est which doubles as the partial-sum accumulator and the LZCS input — see r_recip_est row below).
* - `ac_gte_gpf_scale` (line ~71) covers stage 4's `mtc2 IR0..3 + nop2 + gte_cmdw_gpf + mfc2 MAC1/2/3 + sra`.
* We do NOT call it for the symmetric reason: the normalize in-place semantics overwrite the input regs (r_sx/r_sy/r_sz) with the normalized output,
* which `ac_gte_gpf_scale`'s r_dx/r_dy/r_dz output GPRs would not match.
* `gte_cmdw_sqr` and `gte_cmdw_gpf` primitive macros ARE used in the inlined body, so changes to those primitives
* (e.g., the libgte `fake_cmd` signature bits) propagate automatically. The components remain available for callers that want the explicit GPR-shape variants.
*
* Argument aliasing (9 unique physical regs needed, can drop to 8 with r_sq_y ≡ r_lzcr):
* r_sx, r_sy, r_sz : src components in regs (clobbered by mtc2 → IR1/2/3 in stage 1, then by mfc2 MAC1/2/3 in stage 4 — in-place semantics)
* r_sq_y, r_sq_z : MAC2, MAC3 → DIE after stage 2 accumulate (r_sq_y can alias r_lzcr after stage 2 to save one reg)
* r_recip_est : ≡ r_sqmag — multi-purpose (holds |v|² in stage 2, shift-input in stage 3, sqrtbl[index] in stage 4)
* r_lzcr : LZCR value, alive across stage 3 (srav path needs `24 - LZCR`)
* r_shift : (31 - LZCR & ~1) >> 1 — final srav amount (stages 3-4)
* r_tmp : scratch (shift count, branch target, lookup addr, table base)
*
* GPR ccount peak: 9.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
* Words: ~35 (pending re-gen; matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local. */
/* ─── Binds_NormalizeV3S4 — declared here so the MipsAtom_Proc_ body can reference
* O_(Binds_NormalizeV3S4,*). Inlined at the proc-call site; not exposed in gen/macs.h. */
typedef Struct_(Binds_NormalizeV3S4) {
U4 src; /* V3_S4* (scratch address — read from tape) */
U4 dst; /* V3_S4* (scratch address — write to tape) */
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
typedef Struct_(RegUse_normalize_v3s4_proc) {
Reg scratch; // Scratch base carrier.
Reg src_ptr;
Reg dst_ptr;
Reg recip_est; // |v|² sum + shift-input + sqrtbl[index]
Reg norm; Reg shift;
Reg src_x;
union { Reg mac1_scratch; } t3;
union { Reg mac2_scratch; } t4;
union { Reg shift_count, btarget, lookup_addr, src_z; } t5;
};
/* ─── Full normalize (all 4 stages inline) ───
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav).
*
* Parameterized by caller-provided scratch base + src/dst offsets.
* The caller passes r_src_offset and r_dst_offset as compile-time constants
* (typically derived from O_ macros in the caller's struct schema, e.g., `O_(CallerBundleScratch, fwd)`).
*
* This design lets any caller (with a scratch base + struct schema) use `normalize_v3s4_proc`
* without putting magic offsets in the C-side bundle helper — the offsets come from O_ macros at the call site.
*
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
* r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4
* r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3)
* r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra
* r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y
* r_norm : |v|² sum (stage 2) → half-shift (stage 3) → 1/|v| (stage 4 IR0)
* r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav
* r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr)
*
* Atom_labels are srav_path / aligned_done
* (NOT namespaced — they're internal to this proc;
* the metaprogram's per-atom-name enum emission handles any collision across different atoms/files that share the same labels).
*
* Pool cost: 11 GPRs (well within the 9-10 caller-trash GPR budget when r_scratch is a wave-context carrier).
*
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
* Words: ~59 (matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
*/
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_normalize_v3s4_proc r)
MipsAtom_Proc_(aa, {
add_si(r.src_ptr, r.scratch, src_offset), /* r_src_ptr = &src */
/* NOTE: The bundle-specific scratchpad offset schema was intentionally kept out of this file.
* gte.atom.c is the GENERIC GTE primitives file — it exposes only the parameter-style normalize_v3s4_proc for any future caller. */
I_ void normalize_v3s4_proc(
MipsAtomBuilder_R ab
, U4 r_src /* GPR code: scratch base carrier (wave-context, e.g., R_T4) */
, U4 r_dst /* GPR code: scratch dst pointer carrier (wave-context, e.g., R_T5) */
, U4 r_sx, U4 r_sy, U4 r_sz /* GPR codes: src.x/y/z scratch (atom-local) */
, U4 r_sq_y, U4 r_sq_z /* GPR codes: MAC1/2 scratch (atom-local) */
, U4 r_recip_est /* GPR code: |v|² sum + shift-input + sqrtbl[index] (atom-local) */
, U4 r_lzcr /* GPR code: LZCR value (atom-local) */
, U4 r_shift /* GPR code: final srav amount (atom-local) */
, U4 r_tmp /* GPR code: scratch (shift count, branch target, lookup addr, table base) */
)
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
MipsAtom_Proc_(normalize_v3s4, ab, {
/* ── I/O wrapper (~10 words: 3 bind-pop + 3 src-load + 1 nop + 3 dst-store) ─── */
load_word(r_src, R_TapePtr, O_(Binds_NormalizeV3S4,src)), /* pop src ptr (scratch addr) */
load_word(r_dst, R_TapePtr, O_(Binds_NormalizeV3S4,dst)), /* pop dst ptr (scratch addr) */
add_ui_self( R_TapePtr, S_(Binds_NormalizeV3S4)),
load_word(r_sx, r_src, O_(V3_S4,x)),
load_word(r_sy, r_src, O_(V3_S4,y)),
load_word(r_sz, r_src, O_(V3_S4,z)),
nop, /* load-delay */
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
* r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */
mac_load_v3s4(r.src_x, r.recip_est, r.t5.lookup_addr, r.src_ptr, 0),
/* ── 48-word normalize body (preserved verbatim from ac_normalize_v3s4) ─────── */
// Stage 1: mtc2 src → IR1/2/3, SQR fires (MAC1/2/3 = IR², IR ← MAC saturated)
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop, gte_cmdw_sqr,
// Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS
gte_mv_from_data_r(r_sq_y, C2_MAC1), /* r_sq_y = MAC1 = sx² */
gte_mv_from_data_r(r_sq_z, C2_MAC2), /* r_sq_z = MAC2 = sy² */
gte_mv_from_data_r(r_recip_est, C2_MAC3), /* r_recip_est = MAC3 = sz² */
nop, /* MFC2→GPR load delay (1 slot) */
add_u(r_recip_est, r_recip_est, r_sq_z), /* r_recip_est += sy² */
add_u(r_recip_est, r_recip_est, r_sq_y), /* r_recip_est += sx² (sum = |v|²) */
gte_mv_to_data_r( r_recip_est, C2_LZCS), /* LZCS = |v|² */
nop2,
gte_mv_from_data_r(r_lzcr, C2_LZCR), /* r_lzcr = LZCR (count of leading bits) */
nop, /* MFC2→GPR load delay (1 slot) */
// Stage 3: compute shift amount, align |v|² to bit 24, lookup 1/|v|
and_i( r_lzcr, r_lzcr, -2), /* r_lzcr &= ~1 (force even for halving) */
li_s( r_shift, 31), /* r_shift = 31 */
sub_s( r_shift, r_shift, r_lzcr), /* r_shift = 31 - LZCR */
shift_aright( r_shift, r_shift, 1), /* r_shift = (31 - LZCR) / 2 */
add_si( r_tmp, r_lzcr, -24), /* r_tmp = LZCR - 24 (signed, for branch) */
branch_lt_zero(r_tmp, atom_offset(srav_path, aligned_done)), nop,
jump_rel( atom_offset(aligned_done, srav_path)),
shift_lleft_var(r_recip_est, r_recip_est, r_tmp), /* BD-slot of branch_equal: r_recip_est = |v|² << (LZCR - 24) */
atom_label(srav_path) /* SRAV path: |v|² is small (top bit < bit 24) */
li_s( r_tmp, 24),
sub_s( r_tmp, r_tmp, r_lzcr), /* r_tmp = 24 - LZCR */
shift_aright_var(r_recip_est, r_recip_est, r_tmp), /* r_recip_est = |v|² >> (24 - LZCR) */
atom_label(aligned_done) /* Both paths converge here with |v|² aligned to bit 24 */
/* r_recip_est now holds |v|² aligned to bit 24 — convert to byte offset, -64 to skip zero pad. */
add_si( r_recip_est, r_recip_est, -64),
shift_lleft( r_recip_est, r_recip_est, 1), /* r_recip_est *= 2 (half-word index) */
/* Reference OUR local sqrtbl via &-address split. Compiler/linker resolves both halves. */
load_upper_i( r_tmp, u4_hi(& gte_normalize_sqr_tbl)), /* lui */
or_i_self( r_tmp, u4_lo(& gte_normalize_sqr_tbl)), /* ori */
add_u( r_tmp, r_tmp, r_recip_est), /* r_tmp = sqrtbl base + byte offset (matches libgte 0x80016118: addu t5,t5,t4) */
load_half( r_recip_est, r_tmp, 0), /* r_recip_est = sqrtbl[r_recip_est] = 1/|v| estimate */
nop, /* retire load_half before MTC2 (matches libgte 0x80016120: nop) */
// Stage 4: mtc2 IR0..3, GPF (MAC = IR0*IR), mfc2 MAC, srav finalize
gte_mv_to_data_r(r_recip_est, C2_IR0), /* IR0 = 1/|v| estimate */
gte_mv_to_data_r(r_sx, C2_IR1), /* IR1 = src.x */
gte_mv_to_data_r(r_sy, C2_IR2), /* IR2 = src.y */
gte_mv_to_data_r(r_sz, C2_IR3), /* IR3 = src.z */
nop2, /* COP2 transfer latency (2 slots) */
gte_cmdw_gpf,
gte_mv_from_data_r(r_sx, C2_MAC1), /* MAC1 → r_sx (overwrites src.x with raw reciprocal-scaled) */
gte_mv_from_data_r(r_sy, C2_MAC2),
gte_mv_from_data_r(r_sz, C2_MAC3),
shift_aright_var(r_sx, r_sx, r_shift),
shift_aright_var(r_sy, r_sy, r_shift),
shift_aright_var(r_sz, r_sz, r_shift),
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
/* ── I/O wrapper tail (~3 words) ───────────────────────────────────────────── */
store_word(r_sx, r_dst, O_(V3_S4,x)),
store_word(r_sy, r_dst, O_(V3_S4,y)),
store_word(r_sz, r_dst, O_(V3_S4,z)),
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
add_u_self( r.norm, r.t3.mac1_scratch),
add_u_self( r.norm, r.t4.mac2_scratch),
gte_mv_to_data_r( r.norm, C2_LZCS), LdSlot_ nop2,
gte_mv_from_data_r(r.shift, C2_LZCR), LdSlot_ nop,
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
* After the component: r_shift = even(LZCR), r_norm = half-shift, r_mac1_scratch = |v|². */
mac_lzcr_round_even_half_shift(r.shift, r.norm, r.t3.mac1_scratch),
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
add_si( r.t5.btarget, r.shift, -24),
branch_lt_zero(r.t5.btarget, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
BdSlot_ shift_lleft_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.btarget), /* src=sum (r_mac1_scratch), dst=same */
atom_label(srav_path)
li_s( r.t5.shift_count, 24),
sub_s(r.t5.shift_count, r.t5.shift_count, r.shift),
shift_aright_var(r.t3.mac1_scratch, r.t3.mac1_scratch, r.t5.shift_count), /* src=sum (r_mac1_scratch), dst=same */
atom_label(aligned_done)
// Save the shift count to r_shift before the next 5 instructions overwrite r_norm (the sqrtbl lookup loads 1/|v| into r_norm, which becomes IR0 in stage 4).
or_u(r.shift, r.norm, 0), /* r_shift ← shift count (preserved through stage 4) */
/* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */
add_si( r.t3.mac1_scratch, r.t3.mac1_scratch, -64),
shift_lleft(r.t3.mac1_scratch, r.t3.mac1_scratch, 1),
mac_load_word_imm(r.t5.lookup_addr, & gte_normalize_sqr_tbl), add_u_self(r.t5.lookup_addr, r.t3.mac1_scratch),
load_half(r.norm, r.t5.lookup_addr, 0), /* r_norm = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
/* r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */
LdSlot_ load_word(r.t5.src_z, r.src_ptr, O_(V3_S4,z)), /* r_branch_tmp = src.z (for IR3 in stage 4) */
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_norm = 1/|v|). */
LdSlot_ mac_gte_general_purpose_interopolation(
r.norm,
r.src_x, /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
r.recip_est,
r.t5.src_z, /* IR3 = src.z (reloaded) */
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
LdSlot_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
LdSlot_ nop
),
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
mac_store_v3s4(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
/* ── atom_reads(R_TapePtr) atom_writes(R_TapePtr) ────────────────────────── */
mac_yield()
})
#pragma endregion MACs (Mips Atom Components)
#pragma endregion Atom Procs
#pragma region Bsked Atoms
#pragma region Baked Atoms
typedef Struct_(Binds_SetGteMT3S2S4) {
MT3_S2S4* transform;
+84 -14
View File
@@ -177,20 +177,43 @@ enum {
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
*/
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
gte_shift_sf = 19, gte_width_sf = 1,
gte_shift_mx = 17, gte_width_mx = 2,
gte_shift_v = 15, gte_width_v = 2,
gte_shift_cv = 13, gte_width_cv = 2,
gte_shift_lm = 10, gte_width_lm = 1,
gte_shift_cmd = 0, gte_width_cmd = 6,
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
gte_shift_fake_cmd = 20,
gte_width_fake_cmd = 5,
gte_mask_fake_cmd = 0x1F,
};
/* --- GTE Control Register Aliases (Pitfall 1) ---
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon:
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
* Cross-alias writes inside one atom body, or across the wave-context boundary,
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes
* (CHECK_RULES row) warns about each pair per source. See
* docs/gte_reference.md §"Control-register alias table" for the silicon
* rationale and the libgte outer-product convention.
*/
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
* for the canonical write pattern.
*/
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
* Preprocessor-visible integer ids for the COP2 control register file.
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
@@ -327,13 +350,13 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd ) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) (((x) & gte_mask_fake_cmd) << gte_shift_fake_cmd)
#define enc_gte_sf(sf) ((sf) << gte_shift_sf )
#define enc_gte_mx(mx) ((mx) << gte_shift_mx )
#define enc_gte_v(v) ((v) << gte_shift_v )
#define enc_gte_cv(cv) ((cv) << gte_shift_cv )
#define enc_gte_lm(lm) ((lm) << gte_shift_lm )
#define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
/* Composite: all six GTE fields + the COP2/CO base. */
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
@@ -391,6 +414,45 @@ enum { _C2_TX_SUBS_ = 0
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
* The combine is (pass1 << 3) + pass2. */
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
* Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
* Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
* keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
#define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's
* ApplyMatrixLV when the GTE pipeline expects R*pos >> 12. The shift produces
* values like (-270, 710, 1713) which match the C11 reference path. */
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
/* SQR / GPF cosmetic-bits compat helpers.
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
* The hardware ignores these bits (per PSX-SPX line 48). */
@@ -421,6 +483,14 @@ enum { _C2_TX_SUBS_ = 0
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31)
* down to even. The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1;
* clearing bit 0 ensures the subtraction result is always odd,
* so the >> 1 division is consistent (no 0.5 loss). */
enum {
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
};
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
+156 -43
View File
@@ -23,13 +23,13 @@
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
*
* This behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
* to author and compose programs with. From here various conventions can be further applied.
* To make things easier to understand it may be better to focus on what this ABI does not have.
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
* Branching nearly is always downstream. Stack usage is non-existent.
* Branching nearly is always downstream. Automatic stack usage is non-existent.
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
* In it's current form with the C11 macro dsl, the user also has fullfill manual register allocation per atom.
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
*
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
* or, basically anything from the 5th generation consoles and onward.
@@ -39,10 +39,10 @@
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
* and core atoms to take those newer hardware features into account. For example, you can easily expand
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
* automatic register allocation means the user cannott ignore excessive argument shuffle across workload or
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
*
* Learning data-oreinted code becomes a natural progression. Your not fighting a stack-based procedural
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
@@ -100,17 +100,30 @@ enum {
// S 0-7
};
typedef U2 Reg; // Register parameter used with atom or atom component procedures
typedef U4 const MipsCode; // Underlying type to mips asm words.
typedef Slice_(MipsCode);
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
typedef U4 const MipsAtom;
typedef Slice_(MipsAtom);
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
// MipsAtom* <identifier>[...];
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the blow returns 'MipsAtom'.
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
// Used for atoms with value-args
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
// internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
// expands to:
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; }
#define MipsAtom_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); }
// internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
// The atom name is derived by the Lua metaprogram from the preceding
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
// strips the `_proc` suffix).
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
// MipsAtomComp_(ac_X) { body }
@@ -119,30 +132,30 @@ typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ac_X, ab, { body })
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
// expands to:
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
// MipsCode ac_X[] align_(4) = { body };
// atombuilder_unroll(ab, slice_from_array(MipsCode, ac_X));
// MipsCode atom_comp_code[] align_(4) = { body };
// atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
// }
// The body must NOT include mac_yield() (the parent atom yields).
// Inline-only callers (the generated `mac_<name>` aliases) skip this arg via metaprogram filtering;
// escape callers (ac_<name> invoked as a function) pass a long-lived builder.
#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(ab, slice_from_array(MipsCode, sym)); }
// The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
Files containing only atoms and atom components.
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
The macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
The constant is in `.rodata` and unreferenced; the linker may eliminate it.
The two-level concat + `__LINE__` suffix makes the identifier unique per call site
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
The constant is in `.rodata` so the linker may eliminate it.
Two-level concat + `__LINE__` suffix makes the identifier unique per call site
(identifier embeds the source line, so duplicates across `#include`d files don't collide). */
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
typedef Slice_MipsAtom Tape;
/* The 'Exit' Atom */
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(R_RA), nop };
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
@@ -186,14 +199,14 @@ FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
#define tb_emit_(atom) tb_emit(& tb, atom)
#define tb_data_(field, data) tb_data(& tb, u4_(data))
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), tb->used); tb->used += atoms.len; }
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), S_slice(atoms)); tb->used += atoms.len; }
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
@@ -231,39 +244,139 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
};
#pragma endregion Macro Atom Components
#pragma region Mips Atom Builder
#pragma region Atom Builder
// This helps with runtime procedural authoring of mips atoms.
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
// FArena Related
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
// Whatever the builder is writting to should most likely coresspond
// to something that can fit within instruction cache?
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode code) {
// Usual way to resolve an atom after the bulder is done.
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
assert(ab->capacity - ab->used - code.len);
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot (arena accumulation) */
mem_copy(u4_(dest), u4_(code.ptr), code.len);
mem_bump(ab->start, ab->capacity, & ab->used, code.len);
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
}
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot */
mem_copy(u4_(dest), u4_(ac_yield), S_(ac_yield));
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
}
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
#define mipsatom_from_builder(ab) C_(MipsAtom*, (ab).start)
// tb_emit_builder(tb, ab) — emit the builder's atom into the tape and advance tb->used.
// Thin wrapper around tb_emit(tb, mipsatom_from_builder(ab[0])).
// Equivalent to tb_emit(tb, code_<name>) for runtime-built atoms.
FI_ void tb_emit_builder(TapeBuilder_R tb, MipsAtomBuilder_R ab) { tb_emit(tb, mipsatom_from_builder(ab[0])); }
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
#pragma endregion Mips Atom Builder
#pragma region Atom Arena
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
arena->start = u4_(mem.ptr);
arena->capacity = mem.len;
arena->used = 0;
}
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
assert(aa->capacity - aa->used - code.len);
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
return C_(MipsAtom*, dest);
}
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
#pragma endregion Atom Arena
#pragma region RegFile (Register File Allocator)
// A specialized allocator utilized to help the user track which registers are bound to values
// that must be preserved for the arena's bounds.
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
/* ABI + tape reserves that are never handed out by alloc. */
U4 const regfile_abi_mask =
(1u << R_0) | (1u << R_AT) |
(1u << R_K0) | (1u << R_K1) |
(1u << R_GP) | (1u << R_SP) |
(1u << R_FP) | (1u << R_RA) |
(1u << R_T8) | (1u << R_T9); /* AtomJmp + TapePtr */
typedef Struct_(RegFile) {
A2_U2 GPR;
A2_U2 GTE;
};
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
FI_ void regfile_init(RegFile_R rf) {
/* pack the 32-bit ABI mask into the two U2s */
rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
rf->GTE[0] = rf->GTE[1] = 0;
}
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
typedef Struct_(RegFile_RInfo) {
U2_R section;
U2 mask;
B2 occupied;
};
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
U2 s_id = r_id >> 4;
U2_R section = & file[s_id];
U2 mask = u2_(1u << (r_id & 15));
B2 occupied = (section[0] & mask) != 0;
return (RegFile_RInfo){section, mask, occupied};
}
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
if (info.occupied == false) {
info.section[0] |= info.mask;
result = r_id;
}
return result;
}
I_ Reg regfile_alloc(RegFile_R rf) {
U2 allocated = 0;
for index_iter(Reg, r_id, R_T0, <=, R_T7) {
allocated = regfile__alloc_helper(rf->GPR, r_id); Jmp_nZero_(allocated,resolved);
}
allocated = regfile__alloc_helper(rf->GPR, R_V0); Jmp_nZero_(allocated,resolved);
allocated = regfile__alloc_helper(rf->GPR, R_V1);
assert(allocated != 0);
resolved: return allocated;
}
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
assert(info.occupied == false);
info.section[0] |= info.mask;
return r_id;
}
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
B4 occupied = u4_r(rf->GPR)[0] & mask;
assert(occupied == false);
u4_r(rf->GPR)[0] |= mask;
}
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
if (regfile_abi_mask & mask) return;
u4_r(rf->GPR)[0] &= ~mask;
}
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
/* never free the ABI set */
if (regfile_abi_mask & (1u << r_id)) return;
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
info.section[0] &= ~info.mask;
}
FI_ void regfile_reset(RegFile_R rf) {
rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
}
FI_ void regfile_reset_mask(RegFile_R rf, U4 mask) {
rf->GPR[0] = u4_lo(mask);
rf->GPR[1] = u4_hi(mask);
}
#pragma endregion RegFileArena (Register File Allocator)
#pragma region Mips Atom Procs
#pragma endregion Mips Atom Procs
+19 -11
View File
@@ -9,35 +9,43 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#pragma region MACs (Mips Atom Component)
FI_ Slice_MipsCode ac_load_v2s2(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, {
load_half( rs_x, r_base, O_(V3_S2,x)),
load_half( rs_y, r_base, O_(V3_S2,y)),
// FI_ Slice_MipsCode ac_load_imm
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, {
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_v3s4(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, {
load_word( rs_x, r_base, O_(V3_S4,x)),
load_word( rs_y, r_base, O_(V3_S4,y)),
load_word( rs_z, r_base, O_(V3_S4,z)),
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_load_p3s4 ac_load_v3s4
#define mac_load_p3s4 mac_load_v3s4
FI_ Slice_MipsCode ac_store_v3s4(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, {
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_word(rt_x, base, offset + O_(V3_S4,x)),
store_word(rt_y, base, offset + O_(V3_S4,y)),
store_word(rt_z, base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_store_p3s4 ac_store_v3s4
#define mac_store_p3s4 mac_store_v3s4
FI_ Slice_MipsCode ac_sub_v3s4(MipsAtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, {
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(rds_x, rds_x, rt_x),
sub_s(rds_y, rds_y, rt_y),
sub_s(rds_z, rds_z, rt_z),
})
FI_ Slice_MipsCode ac_store_rects2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, {
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
+11 -5
View File
@@ -24,6 +24,7 @@ enum {
};
typedef Array_(U1, 2);
typedef Array_(U2, 2);
typedef Array_(U4, 2);
typedef Array_(S2, 2);
typedef Array_(S2, 3);
@@ -46,6 +47,9 @@ typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
typedef V3_S4 P3_S4;
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
@@ -64,6 +68,8 @@ typedef Array_(V2_S2, 2);
typedef Array_(V2_S2, 3);
typedef Array_(V2_S2, 4);
#define r1u2(p0,p1) (R1_U2){p0,p1}
enum {
fp_one = (1 << 12),
};
@@ -106,10 +112,10 @@ FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[2] *= b[2];
}
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
+21 -12
View File
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
assert(amount <= (cap - used[0]));
used[0] += amount;
}
@@ -58,13 +58,13 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
typedef Slice_(B1);
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
#define slice_end(slice) ((slice).ptr + (slice).len)
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
#define S_slice(s) ((s).len * S_((s).ptr[0]))
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
@@ -72,23 +72,30 @@ typedef Slice_(B1);
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) }
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
#define slice_zero(s) slice_zero_(slice_to_ut(s))
FI_ void slice_copy_(Slice dest, Slice src) {
assert(dest.len >= src.len);
assert(S_slice(dest) >= S_slice(src));
slice_assert(dest);
slice_assert(src);
mem_copy(dest.ptr, src.ptr, src.len);
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
}
#define slice_copy(dest, src) do { \
static_assert(T_same(dest, src)); \
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
} while(0)
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
assert(len - used[0] - amount);
U4 ptr = start + used[0]; used[0] += amount;
return slice_ut(ptr, amount);
}
typedef Slice_(U1);
typedef Slice_(U4);
#pragma endregion Slice
@@ -98,18 +105,19 @@ typedef Slice_(U4);
typedef Opt_(farena) { U4 alignment, type_width; };
typedef Struct_(FArena) { U4 start, capacity, used; };
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
arena->start = mem.ptr;
arena->start = u4_(mem.ptr);
arena->capacity = mem.len;
arena->used = 0;
}
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
if (amount == 0) { return (Slice){}; }
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
U4 ptr = arena->start + arena->used;
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
return (Slice){ ptr, to_commit };
mem_bump(arena->capacity, & arena->used, to_commit);
return (Slice){ (B1*)ptr, to_commit };
}
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
@@ -117,6 +125,7 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
arena->used -= save_point - arena->start;
}
FI_ U4 farena_save(FArena arena) { return arena.used; }
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
+19 -8
View File
@@ -2,11 +2,22 @@
# include "gen/macs.h"
# include "gen/offsets.h"
# include "bios.h"
# include "mips.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(dst, u4_hi(imm)),
or_i_self( dst, u4_lo(imm)),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
@@ -20,14 +31,14 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
* 6. sp += 8
*/
internal MipsAtom_(mips_flush_icache) {
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
jump_reg(rret_addr), // jr $ra
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
jump_reg(R_RA), // jr $ra
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (BD)
mac_yield(),
};
+46 -38
View File
@@ -136,31 +136,31 @@ enum {
/* Semantic Aliases for MIPS Registers (O32 ABI) */
, rdiscard = R_0 /* Hardwired to 0 */
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
, rret_0 = R_V0 /* Function return value */
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
, rarg_0 = R_A0 /* First function argument */
, rarg_1 = R_A1 /* Second function argument */
, rarg_2 = R_A2 /* Third function argument */
, rarg_3 = R_A3 /* Fourth function argument */
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
, rstatic_1 = R_S1
, rstatic_2 = R_S2
, rstatic_3 = R_S3
, rstatic_4 = R_S4
, rstatic_5 = R_S5
, rstatic_6 = R_S6
, rstatic_7 = R_S7
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
, rstack_ptr = R_SP /* Stack Pointer */
, rret_addr = R_RA /* Return Address (populated by JAL) */
// , rdiscard = R_0 /* Hardwired to 0 */
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
// , rret_0 = R_V0 /* Function return value */
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
// , rarg_0 = R_A0 /* First function argument */
// , rarg_1 = R_A1 /* Second function argument */
// , rarg_2 = R_A2 /* Third function argument */
// , rarg_3 = R_A3 /* Fourth function argument */
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
// , rstatic_1 = R_S1
// , rstatic_2 = R_S2
// , rstatic_3 = R_S3
// , rstatic_4 = R_S4
// , rstatic_5 = R_S5
// , rstatic_6 = R_S6
// , rstatic_7 = R_S7
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
// , rstack_ptr = R_SP /* Stack Pointer */
// , rret_addr = R_RA /* Return Address (populated by JAL) */
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
@@ -259,22 +259,22 @@ enum { _BitOffsets = 0
, SHAMT_SHIFT = 6 /* Shift Amount */
, FC_SHIFT = 0
/* Bit Masks to prevent overflow into adjacent fields */
/* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
* It is NOT a range guard — it is load-bearing for negative branch offsets
* (the metaprogram emits raw signed offsets; the mask truncates them to the
* 16-bit representation the hardware expects). The static analysis
* `immediate_field_width` check validates ranges at build time. */
, OPCODE_MASK = 0x3F
, REG_MASK = 0x1F
, SHAMT_MASK = 0x1F /* Shift Amount */
, FC_MASK = 0x3F
, IMM_MASK = 0xFFFF
};
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
#define enc_imm(imm) (((imm) & IMM_MASK))
#define enc_op(op) ((op) << OPCODE_SHIFT)
#define enc_rs(rs) ((rs) << RS_SHIFT)
#define enc_rt(rt) ((rt) << RT_SHIFT)
#define enc_rd(rd) ((rd) << RD_SHIFT)
#define enc_shamt(shamt) ((shamt) << SHAMT_SHIFT)
#define enc_fc(fc) ((fc) << FC_SHIFT)
#define enc_imm(imm) ((imm) & IMM_MASK)
/* MIPS R-Type Instruction Format (Register-to-Register) */
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
@@ -318,7 +318,10 @@ enum { _BitOffsets = 0
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
#define LdSlot_
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
// #define and_si and_i
@@ -379,6 +382,9 @@ enum { _BitOffsets = 0
*/
#define jump(off) enc_i(op_j, R_0, R_0, (off))
// Annotate an instruction as filling a branch-delay slot.
#define BdSlot_
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
#define jump_rel(off) branch_equal(R_0, R_0, (off))
@@ -411,6 +417,7 @@ enum { _BitOffsets = 0
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
/* --- Arithmetic I-type (immediate) --- */
@@ -453,11 +460,12 @@ enum { _BitOffsets = 0
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
/* nop — sll $0, $0, 0 */
#define nop shift_lleft(rdiscard, rdiscard, 0)
#define nop shift_lleft(R_0, R_0, 0)
#define nop2 nop, nop
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
+17 -15
View File
@@ -11,18 +11,19 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_pad_set_centered_axes(MipsAtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, {
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF),
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF),
store_word( r_scratch, r_state, O_(PadState,axes)),
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
or_i_self( scratch, PadAxis_Centered & 0xFFFF),
// mac_load_word_imm(scratch, PadAxis_Centered),
store_word( scratch, state, O_(PadState,axes)),
})
FI_ Slice_MipsCode ac_pad_set_id_byte(MipsAtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, {
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_ui( r_id, R_0, id_value),
store_byte(r_id, r_state, O_(PadState,id)),
store_byte(r_id, state, O_(PadState,id)),
})
FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, {
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_ui( r_tmp, R_0, pad_status),
store_word(r_tmp, r_state, O_(PadState,status)),
})
@@ -30,7 +31,7 @@ FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state,
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(MipsAtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, {
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
nor_u( r_buttons, r_buttons, R_0),
store_half( r_buttons, r_pad_state, O_(PadState,buttons)),
})
@@ -54,12 +55,12 @@ FI_ Slice_MipsCode ac_pad_store_inverted_buttons(MipsAtomBuilder_R ab, U1 r_butt
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
*
* Register use (atom-local; no wave-context touched):
* R_T0 = raw base (kept throughout; axes loads read raw[4..7] from R_T0)
* R_T1 = state base (kept throughout; all stores go through R_T1)
* R_T2 = raw[0] status (alive across the disc/pending/id dispatch, then dead)
* R_T3 = raw[1] id (alive across the id dispatch, then dead)
* R_T4 = scratch (shifts, compares, immediate loads, store values)
* R_T5 = scratch (parallel lui+ori for the 0x80808080 axes constant + byte-swap target)
* R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
* R_T1 = state base : Kept throughout; all stores go through R_T1.
* R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
* R_T3 = raw[1] id : Alive across the id dispatch, then dead.
* R_T4 = scratch : Shifts, compares, immediate loads, store values.
* R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
*/
enum {
R_PadRaw = R_T0 atom_reg atom_type(U1),
@@ -124,7 +125,8 @@ atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo), /* fills the buttons-load's delay slot (doesn't read R_T4) */
mac_load_word_imm(R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
+9 -9
View File
@@ -36,12 +36,12 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, bios_pad_buffer_size), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, bios_pad_buffer_size), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, bios_init_pad_2), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
call_reg(R_T2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
@@ -62,9 +62,9 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( rtmp_1, rdiscard, bios_start_pad_2), /* $t1 = 0x13 */
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
call_reg(rtmp_2), /* jalr $t2, $ra */
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
call_reg(R_T2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
+1 -1
View File
@@ -83,7 +83,7 @@ typedef Enum_(U1, PadUnknownId) {
typedef Enum_(U4, PadAxisCentered) {
PadAxis_Centered_Hi = 0x8080,
PadAxis_Centered_Lo = 0x8080,
PadAxis_Centered_Word = 0x80808080U,
PadAxis_Centered = 0x80808080U,
};
typedef Enum_(U1, PadDeadZone) {
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
+2
View File
@@ -15,6 +15,8 @@
#define WORD_COUNT(name, count) enum { words_##name = (count) };
WORD_COUNT(nop, 1)
WORD_COUNT(atom_label, 0)
WORD_COUNT(atom_offset, 0)
WORD_COUNT(load_upper_i, 1)
WORD_COUNT(jump_reg, 1)
WORD_COUNT(jump_link, 1)
+13
View File
@@ -0,0 +1,13 @@
#ifdef INTELLISENSE_DIRECTIVES
#pragma once
#endif
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
// Per-phase register allocations resolved by the lua pass.
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
#define R_GpTmp_Code R_V0_Code
-347
View File
@@ -39,350 +39,3 @@ WORD_COUNT(mac_put_disp_env, 5)
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
WORD_COUNT(mac_put_draw_env, 16)
#define mac_resolve_look_at__input_and_sub(...) \
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)) \
, load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)) \
, load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)) \
, add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)) \
, load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtScratch,scratch_base)) \
, add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtScratch)) /* Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation
* column). Reuse r_tmp0/r_tmp1/r_tmp2. Offsets via O_(ResolveLookAtScratch,*). */ \
, load_word(r_tmp0, r_eye_ptr, O_(P3_S4,x)) \
, load_word(r_tmp1, r_eye_ptr, O_(P3_S4,y)) \
, load_word(r_tmp2, r_eye_ptr, O_(P3_S4,z)) \
, nop /* load-delay */ \
, store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,eye.x)) \
, store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,eye.y)) \
, store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye.z)) /* Stage up_in.x/y/z into the scratchpad (atom 2 reads these for the outer
* product with uz). Reuse r_tmp0/r_tmp1/r_tmp2. */ \
, load_word(r_tmp0, r_up_in_ptr, O_(V3_S4,x)) \
, load_word(r_tmp1, r_up_in_ptr, O_(V3_S4,y)) \
, load_word(r_tmp2, r_up_in_ptr, O_(V3_S4,z)) \
, nop /* load-delay */ \
, store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,up_in.x)) \
, store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,up_in.y)) \
, store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in.z)) /* Compute fwd = target - eye. */ \
, load_word(r_tmp0, r_target_ptr, O_(P3_S4,x)) \
, load_word(r_tmp1, r_target_ptr, O_(P3_S4,y)) \
, load_word(r_tmp2, r_target_ptr, O_(P3_S4,z)) \
, load_word(r_tmp3, r_eye_ptr, O_(P3_S4,x)) \
, load_word(R_AT, r_eye_ptr, O_(P3_S4,y)) \
, load_word(R_V0, r_eye_ptr, O_(P3_S4,z)) \
, nop /* load-delay */ \
, sub_u(r_tmp0, r_tmp0, r_tmp3) \
, sub_u(r_tmp1, r_tmp1, R_AT) \
, sub_u(r_tmp2, r_tmp2, R_V0) /* Store fwd.x/y/z (atom 1 reads these as the normalize src). */ \
, store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,fwd.x)) \
, store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,fwd.y)) \
, store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd.z)) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__input_and_sub, 34)
#define mac_resolve_look_at__cross_uz_up_in_to_right(...) \
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_g = &uz */ \
, add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)) /* r_h = &up_in */ \
, add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)) /* r_f = &right (out) */ \
, nop /* Load a (uz).x/y/z into r_a/r_b/r_c. */ \
, load_word(r_a, r_g, O_(V3_S4,x)) \
, load_word(r_b, r_g, O_(V3_S4,y)) \
, load_word(r_c, r_g, O_(V3_S4,z)) \
, nop /* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (hardcoded; reusing the
* body's last two loads is fine because the load-delay slot is the nop
* after the third load, and mtc2 below doesn't read these regs). */ \
, load_word(r_d, r_h, O_(V3_S4,x)) \
, load_word(R_AT, r_h, O_(V3_S4,y)) \
, load_word(R_V0, r_h, O_(V3_S4,z)) \
, nop /* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */ \
, gte_mv_to_data_r(r_a, C2_IR1) \
, gte_mv_to_data_r(r_b, C2_IR2) \
, gte_mv_to_data_r(r_c, C2_IR3) \
, gte_mv_to_data_r(r_d, C2_VXY0) /* D1 = b.x */ \
, gte_mv_to_data_r(R_AT, C2_VZ0) /* D2 = b.y */ \
, gte_mv_to_data_r(R_V0, C2_VXY1) /* D3 = b.z */ \
, nop2 /* MTC2 retirement (CPU→COP2 2-slot delay) */ \
, gte_cmdw_outer_product /* OP fires; MAC1/2/3 = a × b */ /* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */ \
, gte_mv_from_data_r(r_a, C2_MAC1) \
, gte_mv_from_data_r(r_b, C2_MAC2) \
, gte_mv_from_data_r(r_c, C2_MAC3) \
, nop /* MFC2 retirement */ /* Store out.x/y/z to r_f (out ptr = scratch+32). */ \
, store_word(r_a, r_f, O_(V3_S4,x)) \
, store_word(r_b, r_f, O_(V3_S4,y)) \
, store_word(r_c, r_f, O_(V3_S4,z)) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__cross_uz_up_in_to_right, 29)
#define mac_resolve_look_at__cross_uz_ux_to_up(...) \
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_g = &uz */ \
, add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)) /* r_h = &ux */ \
, add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)) /* r_f = &up (out) */ \
, nop /* Load a (uz).x/y/z into r_a/r_b/r_c. */ \
, load_word(r_a, r_g, O_(V3_S4,x)) \
, load_word(r_b, r_g, O_(V3_S4,y)) \
, load_word(r_c, r_g, O_(V3_S4,z)) \
, nop /* Load b (ux).x/y/z into r_d + R_AT/R_V0. */ \
, load_word(r_d, r_h, O_(V3_S4,x)) \
, load_word(R_AT, r_h, O_(V3_S4,y)) \
, load_word(R_V0, r_h, O_(V3_S4,z)) \
, nop /* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */ \
, gte_mv_to_data_r(r_a, C2_IR1) \
, gte_mv_to_data_r(r_b, C2_IR2) \
, gte_mv_to_data_r(r_c, C2_IR3) \
, gte_mv_to_data_r(r_d, C2_VXY0) \
, gte_mv_to_data_r(R_AT, C2_VZ0) \
, gte_mv_to_data_r(R_V0, C2_VXY1) \
, nop2 \
, gte_cmdw_outer_product \
, gte_mv_from_data_r(r_a, C2_MAC1) \
, gte_mv_from_data_r(r_b, C2_MAC2) \
, gte_mv_from_data_r(r_c, C2_MAC3) \
, nop \
, store_word(r_a, r_f, O_(V3_S4,x)) \
, store_word(r_b, r_f, O_(V3_S4,y)) \
, store_word(r_c, r_f, O_(V3_S4,z)) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__cross_uz_ux_to_up, 29)
#define mac_resolve_look_at__normalize_fwd_to_uz(...) \
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,fwd)) /* r_a = &fwd */ \
, add_si(r_b, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_b = &uz */ \
, nop /* Load src.x/y/z from r_a into r_e/r_f/r_i. */ \
, load_word(r_e, r_a, O_(V3_S4,x)) \
, load_word(r_f, r_a, O_(V3_S4,y)) \
, load_word(r_i, r_a, O_(V3_S4,z)) \
, nop /* load-delay */ /* Stage 1: mtc2 src → IR1/2/3, SQR fires. */ \
, gte_mv_to_data_r(r_e, C2_IR1) \
, gte_mv_to_data_r(r_f, C2_IR2) \
, gte_mv_to_data_r(r_i, C2_IR3) \
, nop \
, gte_cmdw_sqr /* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */ \
, gte_mv_from_data_r(r_d, C2_MAC1) \
, gte_mv_from_data_r(r_g, C2_MAC2) \
, gte_mv_from_data_r(r_recip_est, C2_MAC3) \
, nop \
, add_u(r_recip_est, r_recip_est, r_g) \
, add_u(r_recip_est, r_recip_est, r_d) \
, gte_mv_to_data_r(r_recip_est, C2_LZCS) \
, nop2 \
, gte_mv_from_data_r(r_h, C2_LZCR) \
, nop /* Stage 3: compute shift amount, align |v|² to bit 24. */ \
, and_i( r_h, r_h, -2) \
, li_s( r_shift, 31) \
, sub_s( r_shift, r_shift, r_h) \
, shift_aright(r_shift, r_shift, 1) \
, add_si( r_a, r_h, -24) /* r_a = LZCR - 24 (overlapping with r_recip_est; src ptr no longer needed) */ \
, branch_lt_zero(r_a, atom_offset(srav_path_fwd_to_uz, aligned_done_fwd_to_uz)) \
, nop \
, jump_rel(atom_offset(aligned_done_fwd_to_uz, srav_path_fwd_to_uz)) \
, shift_lleft_var(r_recip_est, r_recip_est, r_a) \
, atom_label(srav_path_fwd_to_uz) \
, li_s( r_a, 24) \
, sub_s( r_a, r_a, r_h) \
, shift_aright_var(r_recip_est, r_recip_est, r_a) \
, atom_label(aligned_done_fwd_to_uz) /* r_recip_est holds |v|² aligned to bit 24. */ \
, add_si( r_recip_est, r_recip_est, -64) \
, shift_lleft(r_recip_est, r_recip_est, 1) \
, load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)) \
, or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)) \
, add_u(r_a, r_a, r_recip_est) \
, load_half(r_recip_est, r_a, 0) \
, nop /* Stage 4: GPF + srav finalize. */ \
, gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_e, C2_IR1) \
, gte_mv_to_data_r(r_f, C2_IR2) \
, gte_mv_to_data_r(r_i, C2_IR3) \
, nop2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_e, C2_MAC1) \
, gte_mv_from_data_r(r_f, C2_MAC2) \
, gte_mv_from_data_r(r_i, C2_MAC3) \
, shift_aright_var(r_e, r_e, r_shift) \
, shift_aright_var(r_f, r_f, r_shift) \
, shift_aright_var(r_i, r_i, r_shift) /* Store result.x/y/z to r_b (dst ptr = scratch+16). */ \
, store_word(r_e, r_b, O_(V3_S4,x)) \
, store_word(r_f, r_b, O_(V3_S4,y)) \
, store_word(r_i, r_b, O_(V3_S4,z)) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__normalize_fwd_to_uz, 59)
#define mac_resolve_look_at__normalize_right_to_ux(...) \
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,right)) /* r_a = &right */ \
, add_si(r_b, r_scratch, O_(ResolveLookAtScratch,ux)) /* r_b = &ux */ \
, nop \
, load_word(r_e, r_a, O_(V3_S4,x)) \
, load_word(r_f, r_a, O_(V3_S4,y)) \
, load_word(r_i, r_a, O_(V3_S4,z)) \
, nop \
, gte_mv_to_data_r(r_e, C2_IR1) \
, gte_mv_to_data_r(r_f, C2_IR2) \
, gte_mv_to_data_r(r_i, C2_IR3) \
, nop \
, gte_cmdw_sqr \
, gte_mv_from_data_r(r_d, C2_MAC1) \
, gte_mv_from_data_r(r_g, C2_MAC2) \
, gte_mv_from_data_r(r_recip_est, C2_MAC3) \
, nop \
, add_u(r_recip_est, r_recip_est, r_g) \
, add_u(r_recip_est, r_recip_est, r_d) \
, gte_mv_to_data_r(r_recip_est, C2_LZCS) \
, nop2 \
, gte_mv_from_data_r(r_h, C2_LZCR) \
, nop \
, and_i( r_h, r_h, -2) \
, li_s( r_shift, 31) \
, sub_s( r_shift, r_shift, r_h) \
, shift_aright(r_shift, r_shift, 1) \
, add_si( r_a, r_h, -24) \
, branch_lt_zero(r_a, atom_offset(srav_path_right_to_ux, aligned_done_right_to_ux)) \
, nop \
, jump_rel(atom_offset(aligned_done_right_to_ux, srav_path_right_to_ux)) \
, shift_lleft_var(r_recip_est, r_recip_est, r_a) \
, atom_label(srav_path_right_to_ux) \
, li_s( r_a, 24) \
, sub_s( r_a, r_a, r_h) \
, shift_aright_var(r_recip_est, r_recip_est, r_a) \
, atom_label(aligned_done_right_to_ux) \
, add_si( r_recip_est, r_recip_est, -64) \
, shift_lleft(r_recip_est, r_recip_est, 1) \
, load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)) \
, or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)) \
, add_u(r_a, r_a, r_recip_est) \
, load_half(r_recip_est, r_a, 0) \
, nop \
, gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_e, C2_IR1) \
, gte_mv_to_data_r(r_f, C2_IR2) \
, gte_mv_to_data_r(r_i, C2_IR3) \
, nop2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_e, C2_MAC1) \
, gte_mv_from_data_r(r_f, C2_MAC2) \
, gte_mv_from_data_r(r_i, C2_MAC3) \
, shift_aright_var(r_e, r_e, r_shift) \
, shift_aright_var(r_f, r_f, r_shift) \
, shift_aright_var(r_i, r_i, r_shift) \
, store_word(r_e, r_b, O_(V3_S4,x)) \
, store_word(r_f, r_b, O_(V3_S4,y)) \
, store_word(r_i, r_b, O_(V3_S4,z)) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__normalize_right_to_ux, 59)
#define mac_resolve_look_at__normalize_up_to_uy(...) \
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,up)) /* r_a = &up */ \
, add_si(r_b, r_scratch, O_(ResolveLookAtScratch,uy)) /* r_b = &uy */ \
, nop \
, load_word(r_e, r_a, O_(V3_S4,x)) \
, load_word(r_f, r_a, O_(V3_S4,y)) \
, load_word(r_i, r_a, O_(V3_S4,z)) \
, nop \
, gte_mv_to_data_r(r_e, C2_IR1) \
, gte_mv_to_data_r(r_f, C2_IR2) \
, gte_mv_to_data_r(r_i, C2_IR3) \
, nop \
, gte_cmdw_sqr \
, gte_mv_from_data_r(r_d, C2_MAC1) \
, gte_mv_from_data_r(r_g, C2_MAC2) \
, gte_mv_from_data_r(r_recip_est, C2_MAC3) \
, nop \
, add_u(r_recip_est, r_recip_est, r_g) \
, add_u(r_recip_est, r_recip_est, r_d) \
, gte_mv_to_data_r(r_recip_est, C2_LZCS) \
, nop2 \
, gte_mv_from_data_r(r_h, C2_LZCR) \
, nop \
, and_i( r_h, r_h, -2) \
, li_s( r_shift, 31) \
, sub_s( r_shift, r_shift, r_h) \
, shift_aright(r_shift, r_shift, 1) \
, add_si( r_a, r_h, -24) \
, branch_lt_zero(r_a, atom_offset(srav_path_up_to_uy, aligned_done_up_to_uy)) \
, nop \
, jump_rel(atom_offset(aligned_done_up_to_uy, srav_path_up_to_uy)) \
, shift_lleft_var(r_recip_est, r_recip_est, r_a) \
, atom_label(srav_path_up_to_uy) \
, li_s( r_a, 24) \
, sub_s( r_a, r_a, r_h) \
, shift_aright_var(r_recip_est, r_recip_est, r_a) \
, atom_label(aligned_done_up_to_uy) \
, add_si( r_recip_est, r_recip_est, -64) \
, shift_lleft(r_recip_est, r_recip_est, 1) \
, load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)) \
, or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)) \
, add_u(r_a, r_a, r_recip_est) \
, load_half(r_recip_est, r_a, 0) \
, nop \
, gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_e, C2_IR1) \
, gte_mv_to_data_r(r_f, C2_IR2) \
, gte_mv_to_data_r(r_i, C2_IR3) \
, nop2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_e, C2_MAC1) \
, gte_mv_from_data_r(r_f, C2_MAC2) \
, gte_mv_from_data_r(r_i, C2_MAC3) \
, shift_aright_var(r_e, r_e, r_shift) \
, shift_aright_var(r_f, r_f, r_shift) \
, shift_aright_var(r_i, r_i, r_shift) \
, store_word(r_e, r_b, O_(V3_S4,x)) \
, store_word(r_f, r_b, O_(V3_S4,y)) \
, store_word(r_i, r_b, O_(V3_S4,z)) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__normalize_up_to_uy, 59)
#define mac_resolve_look_at__populate_and_translate(...) \
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)) \
, add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)) /* Compute the 4 scratch pointers in their dedicated GPRs. */ \
, add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)) /* r_pux = &ux */ \
, add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)) /* r_puy = &uy */ \
, add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_puz = &uz */ \
, add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)) /* r_peye = &eye */ \
, nop /* ── m[0] = (S2)ux ── */ \
, load_word(r_tmp0, r_pux, O_(V3_S4,x)) \
, load_word(r_tmp1, r_pux, O_(V3_S4,y)) \
, load_word(r_tmp2, r_pux, O_(V3_S4,z)) \
, nop \
, store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])) \
, store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])) \
, store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])) /* ── m[1] = (S2)uy ── */ \
, load_word(r_tmp0, r_puy, O_(V3_S4,x)) \
, load_word(r_tmp1, r_puy, O_(V3_S4,y)) \
, load_word(r_tmp2, r_puy, O_(V3_S4,z)) \
, nop \
, store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])) \
, store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])) \
, store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])) /* ── m[2] = (S2)uz ── */ \
, load_word(r_tmp0, r_puz, O_(V3_S4,x)) \
, load_word(r_tmp1, r_puz, O_(V3_S4,y)) \
, load_word(r_tmp2, r_puz, O_(V3_S4,z)) \
, nop \
, store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])) \
, store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])) \
, store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])) /* ── Translation column t[i] = R * (-eye) ─────────────────────────────
* pos = -eye: load eye.x/y/z from r_peye, negate via sub_u from R_0. */ \
, load_word(r_tmp0, r_peye, O_(P3_S4,x)) \
, load_word(r_tmp1, r_peye, O_(P3_S4,y)) \
, load_word(r_tmp2, r_peye, O_(P3_S4,z)) \
, nop \
, sub_u(r_tmp0, R_0, r_tmp0) /* pos.x = -eye.x */ \
, sub_u(r_tmp1, R_0, r_tmp1) \
, sub_u(r_tmp2, R_0, r_tmp2) /* mtc2 IR1/2/3 = pos (for MVMVA — input vector registers). */ \
, gte_mv_to_data_r(r_tmp0, C2_IR1) \
, gte_mv_to_data_r(r_tmp1, C2_IR2) \
, gte_mv_to_data_r(r_tmp2, C2_IR3) \
, nop2 /* MVMVA: MAC1/2/3 = R * IR with cv=0 (no TR vector), mx=0 (rotation matrix),
* sf=0 (no shift), v=0 (V0 = IR1/2/3, no far-plane clipping). The pre-set
* rotation matrix is the one set by the preceding set_gte_world atom.
* gte_cmdw_mvmva is parameterless and defaults to cv=0/mx=0/sf=0/v=0. */ \
, gte_cmdw_mvmva \
, nop /* GTE interlock */ /* mfc2 MAC1/2/3 → r_tmp0/r_tmp1/r_tmp2 (sign-extended into 32-bit GPRs).
* MAC1/2/3 hold R*v with no TR add and no perspective divide — exactly the
* 3 distinct world-space translation values we need for t[0..2]. */ \
, gte_mv_from_data_r(r_tmp0, C2_MAC1) \
, gte_mv_from_data_r(r_tmp1, C2_MAC2) \
, gte_mv_from_data_r(r_tmp2, C2_MAC3) \
, nop \
, store_word(r_tmp0, r_look_at, O_(MT3_S2S4,t[0])) \
, store_word(r_tmp1, r_look_at, O_(MT3_S2S4,t[1])) \
, store_word(r_tmp2, r_look_at, O_(MT3_S2S4,t[2])) \
, mac_yield()
WORD_COUNT(mac_resolve_look_at__populate_and_translate, 50)
-30
View File
@@ -8,36 +8,6 @@
#pragma region hello_camera
// --- atom: resolve_look_at__normalize_fwd_to_uz (62 words) ---
#define _atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz 6
#define _atom_offset_aligned_done_fwd_to_uz_srav_path_fwd_to_uz 1
enum {
atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz = _atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz,
atom_offset_aligned_done_fwd_to_uz_srav_path_fwd_to_uz = _atom_offset_aligned_done_fwd_to_uz_srav_path_fwd_to_uz,
};
// --- atom: resolve_look_at__normalize_right_to_ux (62 words) ---
#define _atom_offset_srav_path_right_to_ux_aligned_done_right_to_ux 6
#define _atom_offset_aligned_done_right_to_ux_srav_path_right_to_ux 1
enum {
atom_offset_srav_path_right_to_ux_aligned_done_right_to_ux = _atom_offset_srav_path_right_to_ux_aligned_done_right_to_ux,
atom_offset_aligned_done_right_to_ux_srav_path_right_to_ux = _atom_offset_aligned_done_right_to_ux_srav_path_right_to_ux,
};
// --- atom: resolve_look_at__normalize_up_to_uy (62 words) ---
#define _atom_offset_srav_path_up_to_uy_aligned_done_up_to_uy 6
#define _atom_offset_aligned_done_up_to_uy_srav_path_up_to_uy 1
enum {
atom_offset_srav_path_up_to_uy_aligned_done_up_to_uy = _atom_offset_srav_path_up_to_uy_aligned_done_up_to_uy,
atom_offset_aligned_done_up_to_uy_srav_path_up_to_uy = _atom_offset_aligned_done_up_to_uy_srav_path_up_to_uy,
};
// --- atom: pad_input_cube_rotation (60 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6
File diff suppressed because it is too large Load Diff
+171 -130
View File
@@ -1,7 +1,7 @@
#pragma region Vendors
#include <stdio.h>
#include <stdlib.h>
#include <assert.h>
// #include <assert.h>
// #include "libgpu.h"
// #include "libetc.h"
// #include "libgte.h"
@@ -43,6 +43,7 @@
#pragma region Hello Camera Headers
# include "gen/macs.h"
# include "gen/offsets.h"
# include "gen/auto_reg.h"
#include "hello_camera.h"
#pragma endregion Hello Camera Headers
@@ -51,10 +52,16 @@
#include "hello_camera.atom.c"
#pragma endregion Hello Joypad TUs
enum {
Scratchpad_Loc = 0x1F800000,
};
#define C_scratch(type) C_(type, Scratchpad_Loc)
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
ResolveLookAtArena_Words = 512,
ResolveLookAtArena_Words = 1024,
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
};
typedef Struct_(SMemory) {
PrimitiveArena primitives;
@@ -75,18 +82,11 @@ typedef Struct_(SMemory) {
PadBiosRaw pad_raw[2];
PadState pad[2];
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
U4_V scratchpad; // d-cache
/* resolve_look_at bundle: pre-built atom arena + atom-refs.
* (Task 12.5 fix: moved from file-scope globals to smem fields.
* Task 12.7 fix: dropped the ResolveLookAtScratch struct-as-view; the
* C-side helper uses `& smem.scratchpad[N]` at hardcoded offsets directly.
* Task 12.8 fix: chain atoms use r_scratch + offset internally; no C-side magic offsets anywhere.
* Task 12.11 fix: ResolveLookAtScratch offset schema moved to hello_camera.atom.c — gte.atom.c
* is the GENERIC GTE primitives file and must not know about the resolve_look_at bundle's scratch layout.) */
U4 resolve_look_at_arena[ResolveLookAtArena_Words]; /* ~2 KB; bumped from 420 per Task 4 subagent */
MipsAtom* resolve_look_at_atom_addrs[7];
MipsAtomBuilder resolve_look_at_ab_static;
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
MipsAtom* resolve_look_at_atom_addrs[10];
};
global SMemory smem;
extern SMemory smem;
@@ -104,8 +104,7 @@ I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
}
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
void
resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
// Preconditions: eye != target, up_in not collinear with (target - eye).
V3_S4 right, up, forward;
@@ -130,118 +129,160 @@ resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in)
mul_m3s2_v3s4(look_at, & pos, & off);
trans_m3s2( look_at, & off);
}
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
* Called ONCE from main() before the frame loop.
* After this returns, the smem.resolve_look_at_atom_addrs[] array contains valid MIPS atom pointers
* for the frame-time bundle helper to emit via tb_emit(tb, captured_addr).
*
* 7 atoms are within hello_camera.atom.c:
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
* share the GENERIC normalize_v3s4_proc from gte.atom.c
* 0: resolve_look_at__input_and_sub_proc
* 1: resolve_look_at__normalize_fwd_to_uz_proc
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
* 3: resolve_look_at__normalize_right_to_ux_proc
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
* 4: resolve_look_at__cross_uz_ux_to_up_proc
* 5: resolve_look_at__normalize_up_to_uy_proc
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
* 6: resolve_look_at__populate_and_translate_proc
*
* (gte.atom.c contains only normalize_v3s4_proc — bundle-specific scratch layout is no longer exposed to the GTE primitives file.)
*
* GPR pool per atom: 10 free GPRs (R_T0..R_T3 + R_T5..R_T7 + R_V0 + R_V1 + R_AT).
* R_T4 is reserved as the wave-context carrier (R_ResolveScratch).
*/
internal void resolve_look_at_init(void) {
/* Wrap the static arena in a MipsAtomBuilder. */
MipsAtomBuilder_R ab = & smem.resolve_look_at_ab_static;
ab->start = u4_(smem.resolve_look_at_arena);
ab->capacity = ResolveLookAtArena_Words;
ab->used = 0;
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_atom_addrs));
/* Atom 0: resolve_look_at__input_and_sub — stages eye/up_in into scratchpad,
* computes fwd = target - eye; binds R_ResolveScratch (R_T4) as the wave-context carrier for atoms 1-6.
* The body hardcodes R_AT and R_V0 as eye.y/eye.z temps (the existing sub_u(eye.x, eye.y, eye.z) chain from the prior Task 12.7 design). */
smem.resolve_look_at_atom_addrs[0] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__input_and_sub_proc(ab,
R_T0, /* r_target_ptr (popped from tape) */
R_T1, /* r_eye_ptr (popped from tape) */
R_T2, /* r_up_in_ptr (popped from tape) */
R_ResolveScratch, /* r_scratch (wave-context carrier; popped from tape) */
R_T3, /* r_tmp0 */
R_T5, /* r_tmp1 */
R_T6, /* r_tmp2 */
R_T7); /* r_tmp3 */
U4 pin_mask = regfile_abi_mask | (1 << R_ResolveScratch);
RegFile rf = regfile(pin_mask);
/* Atom 1: resolve_look_at__normalize_fwd_to_uz — src=scratch+0, dst=scratch+16 (HARDCODED in body).
* GPR pool: r_scratch (R_T4 carrier) + 10 body GPRs = 11.
* r_a/r_b (R_T0/R_T1) : src/dst pointers (r_a overlaps r_recip_est after the loads)
* r_e/r_f/r_i (R_T2/R_T3/R_T5) : src.x/y/z → result.x/y/z
* r_d/r_g (R_T6/R_T7) : MAC1/2 scratch (dead after stage 2)
* r_h (R_V0) : LZCR
* r_recip_est (R_V1), r_shift (R_AT) : saved throughout */
smem.resolve_look_at_atom_addrs[1] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__normalize_fwd_to_uz_proc(ab,
R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */
R_T0, R_T1, /* r_a, r_b (src/dst ptrs) */
R_T2, R_T3, R_T5, /* r_e, r_f, r_i (src components) */
R_T6, R_T7, /* r_d, r_g (MAC scratch) */
R_V0, /* r_h (LZCR) */
R_V1, /* r_recip_est */
R_AT); /* r_shift */
/* Atom 2: resolve_look_at__cross_uz_up_in_to_right — a=scratch+16, b=scratch+128,
* out=scratch+32 (HARDCODED in body). GPR pool: r_scratch + 7 body + R_AT + R_V0 = 10. */
smem.resolve_look_at_atom_addrs[2] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__cross_uz_up_in_to_right_proc(ab,
R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */
R_T0, R_T1, R_T2, /* r_a, r_b, r_c (a.x/y/z → out.x/y/z) */
R_T3, /* r_d (b.x) */
R_T5, /* r_f (out ptr = scratch+32) */
R_T6, /* r_g (a ptr = scratch+16) */
R_T7); /* r_h (b ptr = scratch+128) */
/* Atom 3: resolve_look_at__normalize_right_to_ux — src=scratch+32, dst=scratch+48 (HARDCODED). */
smem.resolve_look_at_atom_addrs[3] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__normalize_right_to_ux_proc(ab,
U4 r_target_ptr = regfile_alloc(& rf);
U4 r_eye_ptr = regfile_alloc(& rf);
U4 r_up_in_ptr = regfile_alloc(& rf);
U4 r_tmp0 = regfile_alloc(& rf);
U4 r_tmp1 = regfile_alloc(& rf);
U4 r_tmp2 = regfile_alloc(& rf);
U4 r_tmp3 = regfile_alloc(& rf);
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
R_ResolveScratch,
R_T0, R_T1,
R_T2, R_T3, R_T5,
R_T6, R_T7,
R_V0,
R_V1,
R_AT);
r_target_ptr, r_eye_ptr, r_up_in_ptr,
r_tmp0, r_tmp1, r_tmp2, r_tmp3);
/* Atom 4: resolve_look_at__cross_uz_ux_to_up — a=scratch+16, b=scratch+48, out=scratch+64 (HARDCODED). */
smem.resolve_look_at_atom_addrs[4] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__cross_uz_ux_to_up_proc(ab,
/* === ATOM 1: normalize fwd→uz === */
U2 src_offset = O_(ResolveLookAtScratch, fwd);
U2 dst_offset = O_(ResolveLookAtScratch, uz);
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
});
/* === ATOM 2: cross uz×up_in→right === */
U4 r_a_2 = R_T0;
U4 r_b_2 = R_T1;
U4 r_c_2 = R_T2;
U4 r_d_2 = R_T3;
U4 r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
U4 r_g_2 = R_T6; /* a ptr = scratch+16 */
U4 r_h_2 = R_T7; /* b ptr = scratch+128 */
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
R_ResolveScratch,
R_T0, R_T1, R_T2,
R_T3,
R_T5, /* r_f (out ptr = scratch+64) */
R_T6, /* r_g (a ptr = scratch+16) */
R_T7); /* r_h (b ptr = scratch+48) */
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
/* Atom 5: resolve_look_at__normalize_up_to_uy — src=scratch+64, dst=scratch+80 (HARDCODED). */
smem.resolve_look_at_atom_addrs[5] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__normalize_up_to_uy_proc(ab,
/* === ATOM 3: normalize right→ux === */
src_offset = O_(ResolveLookAtScratch, right);
dst_offset = O_(ResolveLookAtScratch, ux);
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
});
/* === ATOM 4: cross uz×ux→up === */
U4 r_a_4 = R_T0;
U4 r_b_4 = R_T1;
U4 r_c_4 = R_T2;
U4 r_d_4 = R_T3;
U4 r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
U4 r_g_4 = R_T6; /* a ptr = scratch+16 */
U4 r_h_4 = R_T7; /* b ptr = scratch+48 */
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
R_ResolveScratch,
R_T0, R_T1,
R_T2, R_T3, R_T5,
R_T6, R_T7,
R_V0,
R_V1,
R_AT);
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
/* Atom 6: resolve_look_at__populate_and_translate — write look_at->m[][] from ux/uy/uz (computed from r_scratch+offset internally),
then compute translation column t[] = R * (-eye). GPR pool: r_look_at + r_scratch + 4 ptr regs + 3 tmp regs = 9. */
smem.resolve_look_at_atom_addrs[6] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
resolve_look_at__populate_and_translate_proc(ab,
R_T0, /* r_look_at (popped from tape; MT3_S2S4*) */
R_ResolveScratch, /* r_scratch (wave-context carrier) */
R_T1, R_T3, R_T5, R_T7, /* r_pux, r_puy, r_puz, r_peye */
R_T2, R_T6, R_V0); /* r_tmp0, r_tmp1, r_tmp2 */
/* === ATOM 5: normalize up→uy === */
src_offset = O_(ResolveLookAtScratch, up);
dst_offset = O_(ResolveLookAtScratch, uy);
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
src_offset, dst_offset,
RegUse_(normalize_v3s4_proc){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
});
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
U4 r_look_at_6a = R_T0; /* tape pop → look_at* */
U4 r_scratch_6a = R_ResolveScratch;
U4 r_pux_6a = R_T1;
U4 r_puy_6a = R_T3;
U4 r_puz_6a = R_T5;
U4 r_tmp0_6a = R_T2;
U4 r_tmp1_6a = R_T6;
U4 r_tmp2_6a = R_V0;
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
r_look_at_6a, r_scratch_6a,
r_pux_6a, r_puy_6a, r_puz_6a,
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
* the matrix pointer (popped from tape). It does NOT need GPR
* assignment from us — it has its own internal GPR usage.
* We just take its address. */
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
* Uses mac_apply_matrix_lv component macro which internally uses
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
* for the mfc2/store. We pass our GPRs. */
U4 r_scratch_6b = R_ResolveScratch;
U4 r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
U4 r_look_at_6b = R_T0; /* tape pop → look_at* */
U4 r_tmp0_6b = R_T2;
U4 r_tmp1_6b = R_T3;
U4 r_tmp2_6b = R_T5;
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
r_scratch_6b, r_peye_6b, r_look_at_6b,
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
U4 r_look_at_6c = R_T0; /* tape pop → look_at* */
U4 r_scratch_6c = R_ResolveScratch;
U4 r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
U4 r_tmp0_6c = R_T2;
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c, R_T3, R_T4);
/* Sanity check: arena didn't overflow. */
assert(ab->used <= ResolveLookAtArena_Words);
assert(ab.used <= ResolveLookAtArena_Size);
}
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
@@ -262,30 +303,33 @@ I_ void resolve_look_at(
, P3_S4* target
, V3_S4* up_in
){
/* Atom 0: input_and_sub — stages eye/up_in into scratchpad + computes fwd. */
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
tb_data(tb, u4_(target)); /* Binds_ResolveLookAtSub.target (C-side P3_S4*) */
tb_data(tb, u4_(eye)); /* Binds_ResolveLookAtSub.eye (C-side P3_S4*) */
tb_data(tb, u4_(up_in)); /* Binds_ResolveLookAtSub.up_in (C-side V3_S4*) */
tb_data(tb, u4_(smem.scratchpad)); /* Binds_ResolveLookAtScratch.scratch_base */
tb_data(tb, u4_(target));
tb_data(tb, u4_(eye));
tb_data(tb, u4_(up_in));
tb_data(tb, u4_(smem.scratchpad));
}
/* Atoms 1-5: NO tb_data — each chain atom uses r_scratch + hardcoded_offset internally (no tape-data pointers between atoms).
Context carrier R_ResolveScratch (R_T4) is preserved across atoms. */
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
/* Atom 6: populate_and_translate — only output pointer is the matrix destination. */
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
tb_data(tb, u4_(look_at)); /* Binds_ResolveLookAtPopAndTrans.look_at (MT3_S2S4*) */
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
// tb_data(tb, u4_(look_at));
}
}
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf)
{
@@ -298,9 +342,9 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]);
tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]);
// tb_emit_(pad_bios_snapshot);
// tb_data_(raw, & smem.pad_raw[1]);
// tb_data_(state, & smem.pad[1]);
tb_emit_(pad_input_cam);
tb_data_(state, & smem.pad[0]);
@@ -336,17 +380,12 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
A2_S2 p; //???
S4 flag; //????
// Camera Look at
if (1)
{
B4 use_c11_path = false;
if (use_c11_path) {
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
}
// Camera look at (Tape)
if (use_c11_path == false)
{
MT3_S2S4* look_at = & smem.cam.look_at;
P3_S4* eye = & smem.cam.pos;
V3_S4* up_in = & v3s4(0, -fp_one, 0);
tb.used = 0; tb_scope_run(& tb) {
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
}
@@ -455,7 +494,8 @@ GCC_OPTIMIZATION_DISABLE
int main(void)
{
smem = (SMemory){0};
smem.scratchpad = C_(U4_V, 0x1F800000);
// TODO(Ed): remove this field we don't need it in smem.
smem.scratchpad = C_(U4_V, Scratchpad_Loc);
// smem.primitives.used = 0;
// smem.active_buf_id = 0;
smem.cam.pos = v3s4(500, -1000, -1500);
@@ -501,3 +541,4 @@ int main(void)
return 0;
}
GCC_OPTIMIZATION_ENABLE
+2 -2
View File
@@ -25,7 +25,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
#pragma region MACs (Mips Atom components)
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
MipsAtomComp_Proc_(ab, {
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
@@ -36,7 +36,7 @@ MipsAtomComp_Proc_(ac_put_disp_env, ab, {
})
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
MipsAtomComp_Proc_(ab, {
/*
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
* References:
+10625
View File
File diff suppressed because one or more lines are too long
+1
View File
@@ -532,6 +532,7 @@ function build-hello_camera {
$compile_args = @()
$compile_args += $f_debug
$compile_args += ($f_define + 'BUILD_DEBUG')
$compile_args += $f_optimize_none
# $compile_args += $f_optimize_intrinsics
# $compile_args += $f_optimize_size
+213 -3
View File
@@ -515,8 +515,7 @@ local function splice_c_lines(source)
local splice_len = nil
if byte == BYTE_BACKSLASH and source:byte(pos + 1) == BYTE_NEWLINE then
splice_len = 2
elseif byte == BYTE_BACKSLASH and source:byte(pos + 1) == BYTE_CR
and source:byte(pos + 2) == BYTE_NEWLINE then
elseif byte == BYTE_BACKSLASH and source:byte(pos + 1) == BYTE_CR and source:byte(pos + 2) == BYTE_NEWLINE then
splice_len = 3
end
@@ -1314,6 +1313,21 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
},
}
--- GTE control-register alias groups.
--- Aliases within a group write to the same C2 control-register slot on real silicon
--- (the silicon double-maps some C2 slots across multiple PSX SDK / libgte conventions).
--- Aliases across groups write to distinct C2 slots.
---
--- Cross-alias writes inside one atom body, or across the wave-context boundary,
--- silently clobber each other. The `check_gte_cr_alias_writes` check warns about
--- each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
--- for the silicon rationale and the libgte outer-product convention.
M.GTE_CR_ALIAS_GROUPS = {
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
}
-- Operand-class table for the COP2->GPR load-delay check.
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
@@ -1575,6 +1589,8 @@ M.INSTRUCTION_LATENCY = {
["atom_bind"] = 0,
["atom_reads"] = 0,
["atom_writes"] = 0,
["BdSlot_"] = 0,
["LdSlot_"] = 0,
}
-- Default cycle cost for unknown macros.
@@ -1948,6 +1964,52 @@ M.INSTRUCTION_GPR_EFFECTS = {
shift_aright_var = { reads = {2, 3}, writes = {1} },
}
-------------------------------------------------------------------------------
-- IMMEDIATE_FIELD_WIDTHS — maps instruction names to their immediate-argument
-- positions (1-based) and field widths (in bits). Consumed by the
-- `immediate_field_width` static-analysis check. Parallel to
-- INSTRUCTION_GPR_EFFECTS.
--
-- `signed = true` means the field is sign-extended (the value must fit in
-- the signed range). `signed = false` (default) means zero-extended.
-------------------------------------------------------------------------------
M.IMMEDIATE_FIELD_WIDTHS = {
-- CPU I-type immediates: 16-bit signed (addiu/addi/slti sign-extend)
add_ui = { { arg = 3, width = 16, signed = true } },
add_si = { { arg = 3, width = 16, signed = true } },
add_ui_self = { { arg = 2, width = 16, signed = true } },
slt_si = { { arg = 3, width = 16, signed = true } },
slt_ui = { { arg = 3, width = 16, signed = true } },
-- CPU I-type immediates: 16-bit unsigned (andi/ori/xori zero-extend)
and_i = { { arg = 3, width = 16 } },
or_i = { { arg = 3, width = 16 } },
or_i_self = { { arg = 2, width = 16 } },
xor_i = { { arg = 3, width = 16 } },
load_upper_i = { { arg = 2, width = 16 } },
-- Load/store offsets: 16-bit signed
load_word = { { arg = 3, width = 16, signed = true } },
load_half = { { arg = 3, width = 16, signed = true } },
load_half_u = { { arg = 3, width = 16, signed = true } },
load_byte = { { arg = 3, width = 16, signed = true } },
load_byte_u = { { arg = 3, width = 16, signed = true } },
store_word = { { arg = 3, width = 16, signed = true } },
store_half = { { arg = 3, width = 16, signed = true } },
store_byte = { { arg = 3, width = 16, signed = true } },
-- Shift amount: 5-bit unsigned
shift_lleft = { { arg = 3, width = 5 } },
shift_lleft_self = { { arg = 2, width = 5 } },
shift_lright = { { arg = 3, width = 5 } },
shift_aright = { { arg = 3, width = 5 } },
shift_aright_var = { { arg = 3, width = 5 } },
-- Branch offsets: 16-bit signed
branch_equal = { { arg = 3, width = 16, signed = true } },
branch_ne = { { arg = 3, width = 16, signed = true } },
branch_le_zero = { { arg = 2, width = 16, signed = true } },
branch_lt_zero = { { arg = 2, width = 16, signed = true } },
branch_ge_zero = { { arg = 2, width = 16, signed = true } },
branch_gt_zero = { { arg = 2, width = 16, signed = true } },
}
-- Bounded GPR-value rules consumed by the same forward event walk as `INSTRUCTION_GPR_EFFECTS`.
-- A rule describes a literal/constant-producing transform; if its required inputs are not constant, the destination is invalidated rather than carrying a stale value.
-- The lattice is deliberately closed to `{kind = "unknown"}` and `{kind = "constant", value = <U4>}`.
@@ -2095,7 +2157,8 @@ local E_MAC_PREFIX_LEN = 4
--- * Unknown `mac_X` (not in `component_index`): fall back to `word_counts[ident]` if present; otherwise emit one opaque event so the cycle budget accounts for the word.
--- * Marker Tokens (`atom_label(...)` / `atom_offset(...)`): Zero events (they are pure metaprogram hints).
---
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack; a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack;
--- a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
---
--- Pure: reads `body_entry` / `component_index` / `word_counts`. Memoization is the caller's responsibility.
--- Callers wanting `word_events` / `word_event_errors` precomputed for many atoms should memoize them per atom.
@@ -2666,4 +2729,151 @@ function M.project_emission(body_text, component_index, word_counts, components)
})
end
-------------------------------------------------------------------------------
-- find_function_decl_for — backward walk for MipsAtomComp_Proc_ name extraction.
--
-- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
-- is derived from the preceding `FI_ Slice_MipsCode ac_X(args)` function
-- declaration. This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner) or (nil, nil).
-- raw_name — e.g. "ac_load_word_imm"
-- args_inner — e.g. "AtomBuilder_R ab, Reg dst, U4 imm"
--
-- The walk finds the LAST "Slice_MipsCode" before before_pos, then skips
-- whitespace + qualifiers (FI_, atom_dbg_skip, comments) until it finds an
-- ident followed by "(". That ident is the function name; the parens contents
-- are the args.
-------------------------------------------------------------------------------
function M.find_function_decl_for(source, before_pos, slice_mips_code_len)
local search_pos = 1
local last_match = nil
while true do
local found = source:find("Slice_MipsCode", search_pos, true)
if not found or found >= before_pos then break end
last_match = found
search_pos = found + slice_mips_code_len
end
if not last_match then return nil, nil end
local pos = last_match + slice_mips_code_len
while pos < before_pos do
-- skip whitespace
while pos <= #source do
local c = source:sub(pos, pos)
if c == " " or c == "\t" or c == "\n" or c == "\r" then
pos = pos + 1
else
break
end
end
if pos > #source then break end
-- skip line comments
if source:sub(pos, pos + 1) == "//" then
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
pos = pos + 1
goto continue
end
-- skip block comments
if source:sub(pos, pos + 1) == "/*" then
local close = source:find("*/", pos + 2, true)
if not close then break end
pos = close + 2
goto continue
end
-- try to read an ident
local ident, ident_end = M.read_ident(source, pos)
if not ident then break end
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner = M.read_parens(source, next_pos)
if inner then
return ident, inner
end
end
-- ident not followed by "(" — it's a qualifier (FI_, atom_dbg_skip, etc); skip it
pos = ident_end
::continue::
end
return nil, nil
end
-------------------------------------------------------------------------------
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
--
-- After the `sym` arg was dropped from MipsAtom_Proc_, the atom name is
-- derived from the preceding `MipsAtom* X_proc(args)` function declaration.
-- This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner) or (nil, nil).
-- raw_name — e.g. "normalize_v3s4" (the _proc suffix is stripped)
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
--
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips
-- whitespace + qualifiers (internal, I_, FI_, comments) until it finds an
-- ident followed by "(". That ident is the function name (with _proc suffix);
-- the suffix is stripped to get raw_name. The parens contents are the args.
-------------------------------------------------------------------------------
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
local search_pos = 1
local last_match = nil
while true do
-- plain=true: "*" is literal, no escaping needed
local found = source:find("MipsAtom*", search_pos, true)
if not found or found >= before_pos then break end
last_match = found
search_pos = found + mips_atom_ptr_len
end
if not last_match then return nil, nil end
local pos = last_match + mips_atom_ptr_len
while pos < before_pos do
-- skip whitespace
while pos <= #source do
local c = source:sub(pos, pos)
if c == " " or c == "\t" or c == "\n" or c == "\r" then
pos = pos + 1
else
break
end
end
if pos > #source then break end
-- skip line comments
if source:sub(pos, pos + 1) == "//" then
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
pos = pos + 1
goto continue
end
-- skip block comments
if source:sub(pos, pos + 1) == "/*" then
local close = source:find("*/", pos + 2, true)
if not close then break end
pos = close + 2
goto continue
end
-- try to read an ident
local ident, ident_end = M.read_ident(source, pos)
if not ident then break end
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner = M.read_parens(source, next_pos)
if inner then
-- strip the _proc suffix to get the atom name
local proc_suffix = "_proc"
if #ident > #proc_suffix and ident:sub(-#proc_suffix) == proc_suffix then
return ident:sub(1, #ident - #proc_suffix), inner
end
-- no _proc suffix — return as-is
return ident, inner
end
end
-- ident not followed by "(" — it's a qualifier; skip it
pos = ident_end
::continue::
end
return nil, nil
end
return M
+1 -2
View File
@@ -47,8 +47,7 @@ local function find_repo_root()
return root
end
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and
--- `package.cpath` (for `lpeg.dll`).
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
---
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
+84 -51
View File
@@ -4,22 +4,21 @@
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
---
--- User-pinned GPRs (added 2026-08-10): The corpus's `register_alias_registry` is consulted to
--- exclude GPRs the user has pinned via `atom_reg` + `_Code` defs (e.g. wave-context carriers like
--- `R_ResolveScratch = R_T4 atom_reg`). These GPRs are unavailable to EVERY atom's source pool,
--- not just to atoms in the same phase — wave-context carriers are preserved across atoms by the
--- wave-context discipline and must never be reallocated.
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
--- These GPRs are unavailable to EVERY atom's source pool.
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will
--- exclude R_T4 from that atom's pool.
---
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
--- emit `phase_register_clash` as an info finding (no build stop). Should be unreachable after the
--- user-pinning + body-parsing fix above; kept as a defensive safety net.
--- emit `phase_register_clash` as an info finding (no build stop).
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
---
--- Pool exhaustion: if a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
--- emit `phase_register_pool_exhausted` as a build-stopping error.
---
--- @class AutoRegResult
--- @field outputs table[] -- {kind=, path=} entries
--- @field errors table[] -- {line=, msg=} entries (build-stops)
@@ -28,20 +27,57 @@
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- The fixed allocation pool: 10 physical GPRs whose `R_<Sym>_Code` macros exist in mips.h (lines 92-107).
-- Each pool entry is the PHYSICAL GPR ident (R_T0 etc.);
-- `gpr .. "_Code"` resolves to the matching `R_Tn_Code` constant the source code references via `#define R_Load_Code R_T0_Code`.
-- Excluded: R_AT (assembler temp per lottes_tape.h:86), R_T8 (deferred to ac_yield_load pattern),
-- R_T9 (R_TapePtr; owned by the tape runtime).
--- ════════════════════════════════════════════════════════════════════════════
--- THE GPR ALLOCATION POOL — what is allocatable, and (more importantly) WHY
--- ════════════════════════════════════════════════════════════════════════════
---
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
--- It allocates from a FIXED 10-register pool.
--- This comment block makes the inclusion AND exclusion criteria obvious so a reader doesn't have
--- to grep lottes_tape.h + mips.h to understand the design.
---
--- ── WHAT'S IN THE POOL (10 GPRs, all caller-trash per the O32 ABI) ────────
--- R_T0..R_T7 (GPR codes 8..15), R_V0..R_V1 (GPR codes 2..3)
--- The workhorse of every atom body. The uesr should be aware of atom allocation across atoms they chain.
--- If they have a collision it means either they didn't saturate the register file optimally for a phase,
--- or the may have made the workload to large for the run.
---
--- ── WHAT'S NOT IN THE POOL — and WHY (the "obvious exclusions") ────────────
--- R_T9 (GPR code 25) — R_TapePtr, the tape instruction stream pointer.
--- Owned by the tape runtime (in tape_run / tape_run_a02_s07).
--- `rgcc(R_TapePtr)` register-variable ties the C compiler's view to $t9 across the whole tape_run.
--- The auto-reg pass MUST NOT clobber this; doing so would desync the C-side tape pointer from the
--- hardware pointer and crash on the next tape_run.
---
--- R_T8 (GPR code 24) — R_AtomJmp, the atom-jump register used by the 4-word yield handshake.
--- Every `mac_yield()` / `mac_yield_tail` does `load_word R_AtomJmp, R_TapePtr, 0` then
--- `jump_reg R_AtomJmp`. The auto-reg pass MUST NOT clobber this either, or the atom dispatcher breaks.
--- Owned by the tape runtime, same family as R_TapePtr.
---
--- R_AT (GPR code 1) — Assembler temporary. Reserved by the MIPS O32 ABI for pseudoinstruction expansion
--- (lottes_tape.h:86, mips.h:93). The ISA's psuedo instructions use it as a scratch temporary.
---
--- R_A0..A3 (codes 4..7) — Function arguments. Used in tape_run_a02_s07, see below.
--- R_S0..S7 (codes 16..23) — Callee-saved. Preserved across C-ABI calls by convention.
--- The `tape_run_a02_s07` variant clobbers them deliberately, but the default `tape_run` does NOT.
--- Kept out of POOL to preserve the conservative default.
--- Add them in a separate "big clobber" pool if/when needed.
---
--- R_K0/K1 (codes 26..27) — Kernel / interrupt handler reserves. Never touched by user code; OS-internal.
--- R_GP/SP/FP/RA (codes 28..31) — Stack frame + return-address. Owned by the C compiler; never allocatable.
--- R_0 (code 0) — Hardwired zero. Cannot be written.
---
local POOL = {
"R_T0", "R_T1", "R_T2", "R_T3",
"R_T4", "R_T5", "R_T6", "R_T7",
"R_V0", "R_V1",
}
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident
-- in POOL. The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines
-- (mips.h:92-123). Only the POOL entries matter for auto_reg — non-pool aliases are out of scope.
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
-- Only the POOL entries matter for auto_reg — non-pool aliases
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
local INT_CODE_TO_POOL_GPR = {
[2] = "R_V0", [3] = "R_V1",
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
@@ -71,8 +107,9 @@ local function allocate_phase(phase_label, decls)
if not next_gpr then
errors[#errors + 1] = {
line = 0,
msg = string.format(
"phase_register_pool_exhausted: phase '%s' requested symbol '%s' but the pool has no remaining registers (max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
msg = string.format("phase_register_pool_exhausted: "
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
.. "(max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
, phase_label, sym),
}
return result, errors
@@ -83,21 +120,17 @@ local function allocate_phase(phase_label, decls)
end
-- Build two projections from corpus.register_alias_registry:
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally
-- -- (wave-context carriers, file-scope pinned aliases)
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
-- Both projections are derived from the same set of entries: every AliasEntry in
-- register_alias_registry has `has_atom_reg = true` (only those entries are added to the
-- registry; see passes/scan_source.lua parse_enum_entry). Each entry's `code` is the integer
-- MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored — they
-- don't affect the auto_reg pool, and they're already excluded from POOL above.
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
local function build_user_pins(corpus)
local user_pinned = {}
local alias_to_gpr = {}
if not corpus.register_alias_registry then
return user_pinned, alias_to_gpr
end
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do
if alias_entry.has_atom_reg and alias_entry.code then
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code]
@@ -111,10 +144,10 @@ local function build_user_pins(corpus)
end
-- Find every physical GPR referenced in the atom body, via EITHER:
-- (a) a hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
-- (b) an alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
-- Returns { [physical_gpr_ident] = count }. The clash-detection and source-pool-exclusion logic
-- only needs the presence of each GPR (boolean test), but keeping the count preserves the
-- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
-- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
-- Returns { [physical_gpr_ident] = count }. Clash-detection and source-pool-exclusion logic
-- only needs the presence of each GPR (boolean test), but keeping count preserves the
-- original find_hardcoded_rn shape so callers can switch without churn.
-- The alias pattern is sorted lexicographically to keep the regex deterministic.
local function find_used_gprs(body_text, alias_to_gpr)
@@ -190,12 +223,10 @@ function M.run(ctx)
end
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in
-- hello_camera.atom.c) MUST NOT be allocated to any auto-reg marker — they're
-- preserved across atoms by the wave-context discipline. The corpus's
-- register_alias_registry is the source of truth for these opt-in pins.
-- Body references to those aliases (via alias_to_gpr) are also excluded on a
-- per-atom basis in step 2 below.
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
local user_pinned, alias_to_gpr = build_user_pins(corpus)
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
@@ -213,8 +244,8 @@ function M.run(ctx)
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
-- Otherwise, allocate a private pool for the atom.
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)` in the atom's `atom_info` line).
-- Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
local atom_name_to_phase = {}
for phase_label, entry in pairs(corpus.atom_phases or {}) do
for _, atom_name in ipairs(entry.atoms or {}) do
@@ -229,8 +260,7 @@ function M.run(ctx)
-- (a) every GPR already committed (phase allocations + prior atom allocations)
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
-- (the latter resolved via alias_to_gpr; this catches cases where the user
-- wrote R_ResolveScratch instead of R_T4 directly)
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
@@ -259,7 +289,8 @@ function M.run(ctx)
if not next_gpr then
errors[#errors + 1] = {
line = 0,
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' but no free registers remain in its scope pool."
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
.. "but no free registers remain in its scope pool."
, atom_scope, sym),
}
else
@@ -271,10 +302,10 @@ function M.run(ctx)
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>). An auto-reg allocation that
-- matched an existing body reference would be impossible by construction. This warning is kept
-- as a defensive safety net for cases the body scanner might miss (e.g. macros that expand to
-- register references the scanner cannot resolve).
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
-- This warning is kept as a defensive safety net for cases the body scanner might miss
-- (e.g. macros that expand to register references the scanner cannot resolve).
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
for atom_scope, decls in pairs(atom_allocations) do
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
@@ -284,7 +315,8 @@ function M.run(ctx)
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
warnings[#warnings + 1] = {
line = atom.line or 0,
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
, atom_scope, allocated_gpr, sym, allocated_gpr),
}
end
@@ -300,7 +332,8 @@ function M.run(ctx)
for _, src in ipairs(sources) do
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable, which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do
per_dir_mappings[sym] = gpr
+45 -70
View File
@@ -3,9 +3,12 @@
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
---
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)`, `MipsAtomComp_Proc_(ac_X, { body })`, and `MipsAtom_Proc_(X, ab, { body })` declarations,
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations (kind="comp_bare" / "comp_proc"),
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
---
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
---
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
--- The directory itself is the namespace, so the filename does not repeat the module name.
@@ -76,7 +79,7 @@ local MACS_FILENAME = "macs.h"
--- @field args string|nil -- Function-args string (function form only)
--- @field line integer -- Source line of the declaration
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
--- @field kind string -- "comp_bare" | "comp_proc" | "atom_proc"
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component — see `project_components`)
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
-- ════════════════════════════════════════════════════════════════════════════
@@ -93,48 +96,21 @@ local M = {}
-- so this file reads it forward rather than re-walking the source.
-- ════════════════════════════════════════════════════════════════════════════
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation.
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
---
--- Convention: function form is
--- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
--- We then verify the preceding context ends with `Slice_MipsCode`
--- (the function-decl keyword with possible qualifiers between).
--- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
--- and the args both come from the preceding `FI_ Slice_MipsCode ac_X(args)`
--- declaration. The shared `duffle.find_function_decl_for` helper does the
--- backward walk; this function returns just the args.
---
--- @param source string
--- @param name string
--- @param name string (retained for signature stability; unused — the walk derives the name)
--- @param before_pos integer
--- @return string|nil
local function find_function_args_for(source, name, before_pos)
-- Find the LAST occurrence of `name + "("` in `source[1..before_pos]`.
local name_open = name .. "("
local last_idx = nil
local scan_pos = 1
while true do
-- Pass `before_pos + 1` so string.find only returns positions < before_pos + 1
-- (string.find's 4th arg `plain` is true; we use the 3rd arg `init` for the upper bound).
local found = source:find(name_open, scan_pos, true)
if not found or found >= before_pos then break end
last_idx = found
scan_pos = found + #name_open
end
if not last_idx then return nil end
-- Verify the preceding context ends with "MipsAtom" (with possible qualifiers between).
local before = source:sub(1, last_idx - 1)
local trimmed = duffle.trim(before)
if trimmed:sub(-#MIPS_ATOM) ~= MIPS_ATOM then
-- Preceding context is not a function declaration.
return nil
end
local open_paren = last_idx + #name -- position of "("
-- scan: MipsAtom ac_X(
local inner = duffle.read_parens(source, open_paren)
-- scan: MipsAtom ac_X(<args>)
if not inner then return nil end
return inner
local _, args_inner = duffle.find_function_decl_for(source, before_pos, #MIPS_ATOM)
return args_inner
end
-- ════════════════════════════════════════════════════════════════════════════
@@ -200,16 +176,17 @@ end
local function project_components(source, scan)
local out = {}
for _, a in ipairs(scan.atoms) do
if a.kind == "comp_bare" or a.kind == "comp_proc" or a.kind == "atom_proc" then
-- `MipsAtom_Proc_` atoms have no `FI_ Slice_MipsCode ac_X(...)` function-decl prelude
-- (the macro sits inside a wrapping `I_ void <proc_name>(...)` body), so the function-args
-- lookup is meaningless; signature defaults to `...` (variadic-ignored).
-- The `mac_<name>` alias expansion discards the `ab` (atom-builder) arg the same way
-- `MipsAtomComp_Proc_` components do.
local args = nil
if a.kind ~= "atom_proc" then
args = find_function_args_for(source, a.raw_name, a.ident_pos)
end
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
if a.kind == "comp_bare" or a.kind == "comp_proc" then
-- Function-args lookup is meaningful for `MipsAtomComp_Proc_` components
-- (the macro sits inside `FI_ Slice_MipsCode ac_X(...)`); the alias expansion
-- discards the `ab` (atom-builder) arg the same way both forms do.
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
-- The pass reads `declaration_comment` directly.
local comment = a.declaration_comment or ""
@@ -221,7 +198,7 @@ local function project_components(source, scan)
body_tokens = a.body_tokens,
args = args,
comment = comment,
kind = a.kind, -- "comp_bare" | "comp_proc" | "atom_proc"; provenance emitter reads this.
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
debug_skip = a.debug_skip == true,
}
end
@@ -400,8 +377,7 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
end
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
--- calls in the component body that target `R_PrimCursor` (these are the
--- RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
--- calls in the component body that target `R_PrimCursor` (these are the RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
--- @param name string
--- @param comp_by_name table<string, Component>
@@ -484,10 +460,9 @@ end
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies
--- don't reference `ab` (the builder is only consumed by the procedural `atombuilder_unroll` line
--- that `MipsAtomComp_Proc_` appends after the body). Inline callers therefore don't need to thread
--- a builder context.
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies don't reference `ab`
--- (the builder is only consumed by the procedural `atombuilder_unroll` line that `MipsAtomComp_Proc_` appends after the body).
--- Inline callers therefore don't need to thread a builder context.
--- @param args_str string|nil
--- @return string
local function signature_from_args(args_str)
@@ -544,7 +519,7 @@ local function build_component_lines(c, counts)
-- Marker comment: emitted once for every skipped component.
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
-- This pass projects `c.debug_skip` and emits the marker as a generated comment.
if c.debug_skip then
lines[#lines + 1] = "/* atom_dbg_skip */"
end
@@ -578,8 +553,8 @@ end
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
--- @param dir string -- the absolute source directory
--- @param sources SourceFile[] -- sources contributing to this directory (for the header comment)
--- @param dir string -- Absolute source directory
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
--- @return string[]
local function header_boilerplate(dir, sources)
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" }
@@ -610,9 +585,9 @@ end
--- Compute the per-directory output path for `.macs.h`.
--- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
--- The directory name is the namespace; the filename does not repeat it.
--- @param dir string -- the absolute source directory
--- @return string -- the output directory
--- @return string -- the full output path
--- @param dir string -- Absolute source directory
--- @return string -- Output directory
--- @return string -- Full output path
local function compute_macs_h_path(dir)
local out_dir = dir .. "/" .. GEN_SUBDIR
local out_path = out_dir .. "/" .. MACS_FILENAME
@@ -622,11 +597,11 @@ end
--- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
--- @param ctx PassCtx
--- @param dir string -- the absolute source directory
--- @param sources SourceFile[] -- sources contributing to this directory (for the header comment)
--- @param components Component[] -- aggregated components from all sources in this directory
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
--- @return string|nil -- path to the written file (nil if no components)
--- @param dir string -- Absolute source directory
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
--- @param components Component[] -- Aggregated components from all sources in this directory
--- @param counts table<string, integer> -- Precomputed word counts (from count_all_components)
--- @return string|nil -- Path to the written file (nil if no components)
local function emit_component_macros_h(ctx, dir, sources, components, counts)
if #components == 0 then return nil end
local out_dir, out_path = compute_macs_h_path(dir)
@@ -665,11 +640,11 @@ local function update_canonical_word_counts(corpus, components, counts)
end
--- @class ComponentDef
--- @field name string -- bare name (without ac_/mac_ prefix)
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc" | "atom_proc"
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- @field name string -- Bare name (without ac_/mac_ prefix)
--- @field line integer -- Definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
--- @field path string -- Absolute source path of the definition
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component)
--- @field debug_skip boolean -- Mirror of the scanner-owned `a.debug_skip`; consumers read this directly
--- (internal) Populate `corpus.components` with this source's components-by-name map.
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
+12 -11
View File
@@ -703,9 +703,9 @@ end
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
--- is the k-th word's source line within the component body.
---
--- @param corpus table -- the corpus from `ctx.shared.corpus`
--- @param corpus table -- From `ctx.shared.corpus`
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
--- @return table[] -- list of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
--- @return table[] -- List of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
local function build_atom_table(corpus, addrs)
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
local atoms_by_name = corpus.atoms_by_name or {}
@@ -834,10 +834,10 @@ end
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
---
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level statements (each entry is a single `load_*` call or other statement).
--- @param body_tokens table[] -- the atom's pre-tokenized body statements (from atom.body_tokens)
--- @param binds_name string -- expected Binds_X name (skip pairs with mismatching binds)
--- @param registries table -- merged registries from collect_per_source_registries
--- @return table[] -- list of {reg = <MIPS index>, field = <field name>}
--- @param body_tokens table[] -- The atom's pre-tokenized body statements (from atom.body_tokens)
--- @param binds_name string -- Expected Binds_X name (skip pairs with mismatching binds)
--- @param registries table -- Merged registries from collect_per_source_registries
--- @return table[] -- List of {reg = <MIPS index>, field = <field name>}
local function parse_body_load_pairs(body_tokens, binds_name, registries)
local pairs = {}
local reg_index_by_name = (registries and registries.register_alias_registry) or {}
@@ -880,9 +880,9 @@ end
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
---
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
--- @param corpus table -- the corpus from `ctx.shared.corpus`
--- @param atom_table table[] -- the cross-ref'd atom table from build_atom_table
--- @param registries table -- merged registries from collect_per_source_registries
--- @param corpus table -- From `ctx.shared.corpus`
--- @param atom_table table[] -- Cross-ref'd atom table from build_atom_table
--- @param registries table -- Merged registries from collect_per_source_registries
--- @return table, table -- (rbind_atoms, rbind_structs)
local function parse_rbind_atoms(corpus, atom_table, registries)
registries = registries or {}
@@ -944,7 +944,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
binds = ai.binds,
fields = struct.fields, -- {name, offset} from scan.binds
bytes = struct.bytes,
regs = pairs, -- ordered list of {reg, field}
regs = pairs, -- Ordered list of {reg, field}
info_line = ai.info_line,
}
table.insert(struct.atom_names, atom_name)
@@ -1780,7 +1780,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type. The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type.
-- The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
local ptr_void_offset = void_chain_offset + 8
+8
View File
@@ -4,9 +4,17 @@
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
---
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
--- The directory itself is the namespace; the filename does not repeat the module name.
---
--- (Task 12.16 note: atom-namespaced enum names — e.g., `atom_offset__normalize_v3s4__srav_path__aligned_done` —
--- were considered to prevent cross-atom label collisions, but the C-side `atom_offset(F, T)` macro in
--- `code/duffle/dsl.atom.h` doesn't know the current atom_name at expansion time, so any namespacing
--- on the metaprogram side breaks the C build. Reverted. The C-side would need a per-atom
--- `CURRENT_ATOM` #define (set by `MipsAtom_`/`MipsAtom_Proc_` macros) plus an updated `atom_offset`
--- macro that uses it. That's a coordinated refactor — deferred to a future track.)
---
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
-- ════════════════════════════════════════════════════════════════════════════
+21 -4
View File
@@ -137,6 +137,16 @@ local QUALIFIER_KEYWORDS = {
local AC_PREFIX = "ac_"
local AC_PREFIX_LEN = 3
-- The function-decl keyword that precedes a MipsAtomComp_Proc_ call.
-- Used by the backward walk in duffle.find_function_decl_for.
local SLICE_MIPS_CODE = "Slice_MipsCode"
local SLICE_MIPS_CODE_LEN = #SLICE_MIPS_CODE
-- The return type that precedes a MipsAtom_Proc_ function declaration.
-- Used by the backward walk in duffle.find_atom_proc_decl_for.
local MIPS_ATOM_PTR = "MipsAtom*"
local MIPS_ATOM_PTR_LEN = #MIPS_ATOM_PTR
--- Strip the "ac_" prefix from a component name.
--- Returns the input unchanged if it doesn't start with the prefix.
--- @param raw_name string
@@ -1357,7 +1367,11 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end
local raw_name = inner:match("^%s*([%w_]+)") or "?"
-- The component name is derived from the preceding function declaration
-- (`FI_ Slice_MipsCode ac_X(...)`), not from the first macro arg (which
-- is now `ab`). The backward walk finds the function decl before open_paren.
local raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
if not raw_name then raw_name = "?" end
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
@@ -1398,9 +1412,12 @@ local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end
-- The atom name is the FIRST ident of the args (matches MipsAtomComp_Proc_'s "first ident" rule).
-- MipsAtom_Proc_ has no `ac_` prefix; `strip_ac_prefix` is a no-op for unprefixed names.
local raw_name = inner:match("^%s*([%w_]+)") or "?"
-- The atom name is derived from the preceding function declaration
-- (`internal MipsAtom* X_proc(...)`), not from the first macro arg (which
-- is now `aa`). The backward walk finds the function decl before open_paren
-- and strips the `_proc` suffix.
local raw_name = duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
if not raw_name then raw_name = "?" end
local name = strip_ac_prefix(raw_name)
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
+280 -3
View File
@@ -39,8 +39,8 @@
--- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts.
---
--- The structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`, `control_transfer_delay_slot_use`) skip atoms/components with `debug_skip == true`.
--- The `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
--- Flagging them as "missing mac_yield" or "BD slot is redundant" is signal noise, not a logic failure.
--- `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
--- Flagging them as "missing mac_yield" or "BD slot is redundant".
--- Other checks (transfer_hazards, gpu_portstore_shape, abi_handoff, enum_alias_membership, …) still apply to debug_skip declarations because real hazards / typos can still surface in them.
---
--- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table:
@@ -452,7 +452,7 @@ local function is_cop2_consumer_of(consumer_event, destination, producer_rel)
end
-- True iff `consumer_event` reads the GPR operand at any position the destination register occupies.
-- The read-position lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
-- read_pos lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
local function is_gpr_consumer_of(consumer_event, destination)
local consumer_token = consumer_event.encoder or consumer_event.ident
local read_pos = duffle.OPERAND_READ_POSITIONS or {}
@@ -2388,6 +2388,274 @@ end
-- ════════════════════════════════════════════════════════════════════════════
-- ════════════════════════════════════════════════════════════════════════════
-- GTE control-register alias + RT-diagonal + TR-naming helpers and checks
-- ════════════════════════════════════════════════════════════════════════════
--- Resolve a `gte_cr_<Alias>` ident to its alias-group entry, or nil if the alias
--- is in a distinct-slot group (or the alias name is not a known C2 control-register alias).
--- Reads `M.GTE_CR_ALIAS_GROUPS` from `duffle.lua`.
local function find_alias_pair_for(alias_name, duffle)
local groups = (duffle and duffle.GTE_CR_ALIAS_GROUPS) or {}
for _, group in ipairs(groups) do
for _, name in ipairs(group[2] or {}) do
if name == alias_name then return group end
end
end
return nil
end
-- True iff `c` (a TokClass entry) is a CPU→COP2 control-register transfer
-- (`gte_mv_to_ctrl_r` / `gte_mv_from_ctrl_r`).
local function is_ctrl_r_transfer(c)
if c == nil then return false end
return c.ident == "gte_mv_to_ctrl_r" or c.ident == "gte_mv_from_ctrl_r"
end
-- Resolve a token's source line. The per-token `line` is the body-relative
-- line; `atom.line` is the source line of the atom declaration; `line_in_body`
-- (atom.paths) maps a body-relative line to its source line. The arithmetic
-- `atom.line + line_in_body[tok.rel] - 1` matches the convention used by
-- check_abi_handoff and check_control_transfer_delay_slot_use elsewhere.
local function atom_body_token_source_line(atom, token, line_in_body)
if line_in_body == nil or token == nil or token.rel == nil then
return atom.line or 0
end
local body_line = line_in_body[token.rel]
if body_line == nil then return atom.line or 0 end
return (atom.line or 0) + body_line - 1
end
-- Check #N: gte_cr_alias_writes
-- Fires one warning per atom per alias-group when the atom body touches two
-- distinct aliases from the same group. Aliases within a group write to the
-- same C2 control-register slot on real silicon; cross-alias writes inside
-- one atom body silently clobber each other.
--
-- Severity: warning. Build continues. The libgte outer-product convention
-- uses only RT-row aliases (which are NOT in `M.GTE_CR_ALIAS_GROUPS`), so
-- the canonical convention does not trigger this check.
local function check_gte_cr_alias_writes(atom, pipe_ctx, findings)
local groups = pipe_ctx.gte_cr_alias_groups or {}
if not next(groups) then return end
local tokens = atom.paths and atom.paths.tokens or {}
local tc = atom.paths and atom.paths.tok_class or {}
local line_in_body = atom.paths and atom.paths.line_in_body
if not next(tokens) then return end
-- Build a per-group set of (alias, source_line) pairs touched in this atom body.
-- Walks every token; when the token is a ctrl-r transfer, the alias is at
-- position tok_idx + 2 (rt, alias, [imm-or-arg]). The pre-classified
-- `tc` table tells us whether the token is a ctrl-r transfer and what its
-- source line is.
local touched = {}
for tok_idx, token in ipairs(tokens) do
local c = tc[tok_idx]
if is_ctrl_r_transfer(c) and tokens[tok_idx + 2] then
local alias = tokens[tok_idx + 2].tok
local group = find_alias_pair_for(alias, pipe_ctx.duffle)
if group then
touched[group[1]] = touched[group[1]] or {}
touched[group[1]][#touched[group[1]] + 1] = {
alias = alias,
line = atom_body_token_source_line(atom, token, line_in_body),
}
end
end
end
-- Fire one warning per group touched with 2+ distinct aliases.
for slot, hits in pairs(touched) do
local seen = {}
local distinct = {}
for _, h in ipairs(hits) do
if not seen[h.alias] then
seen[h.alias] = true
distinct[#distinct + 1] = h
end
end
if #distinct >= 2 then
local aliases = {}
for _, d in ipairs(distinct) do aliases[#aliases + 1] = d.alias end
findings[#findings + 1] = {
atom = atom.name or "",
line = distinct[1].line,
check = "gte_cr_alias_writes",
kind = "warning",
msg = string.format(
"atom '%s' touches %d aliases that share C2[%d]: %s; verify the intent"
, atom.name or "", #distinct, slot, table.concat(aliases, ", ")),
}
end
end
end
-- Check #N+1: rtdiagonal_completeness
-- Fires one info per atom body when the bare `gte_cmdw_mvmva` macro is used.
-- The bare macro encodes only the cmd field; the canonical libgte-2-pass
-- shape uses `gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012` (gte.h:430).
--
-- Severity: info by default. Escalates to warning when
-- `GTE_RT_DIAGONAL_STRICT=1` env var is set (CI / production builds).
--
-- The bare macro IS the right call for the canonical libgte outer-product
-- convention, so this is an opt-out hint rather than a hard warning.
local function check_rtdiagonal_completeness(atom, _pipe_ctx, findings)
local tokens = atom.paths and atom.paths.tokens or {}
local tc = atom.paths and atom.paths.tok_class or {}
local line_in_body = atom.paths and atom.paths.line_in_body
if not next(tokens) then return end
local strict = os.getenv("GTE_RT_DIAGONAL_STRICT") == "1"
for tok_idx, token in ipairs(tokens) do
local c = tc[tok_idx]
if c and c.ident == "gte_cmdw_mvmva" then
findings[#findings + 1] = {
atom = atom.name or "",
line = atom_body_token_source_line(atom, token, line_in_body),
check = "rtdiagonal_completeness",
kind = strict and "warning" or "info",
msg = string.format(
"atom '%s' uses the bare gte_cmdw_mvmva macro; "
.. "the canonical libgte-2-pass shape is gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012 "
.. "(gte.h:430). The bare macro does not encode RT23/RT31/RT32/RT33; "
.. "for a full 3x3 matrix, use the dedicated literal or hand-build via enc_gte_*()."
, atom.name or ""),
}
end
end
end
-- Check #N+2: gte_cr_TR_naming
-- Fires one info per atom body when a `gte_cr_TR[XYZ]` alias is used.
-- Translation-vector registers are the only 3-letter-suffix C2 aliases
-- (`TRX/TRY/TRZ`); an agent who reads `TRX` might typo it as `RT_X` or
-- `RTX0` and either get a compile error (best case) or a build that
-- links but routes the `ctc2` write to the wrong C2 slot.
--
-- Severity: info. The convention is correct; this is a documentation-pointer check.
local function check_gte_cr_TR_naming(atom, _pipe_ctx, findings)
local tokens = atom.paths and atom.paths.tokens or {}
local tc = atom.paths and atom.paths.tok_class or {}
local line_in_body = atom.paths and atom.paths.line_in_body
if not next(tokens) then return end
local touched = false
local first_line = 0
for tok_idx, token in ipairs(tokens) do
local c = tc[tok_idx]
if c and c.ident and c.ident:match("^gte_cr_TR[XYZ]$") then
touched = true
if first_line == 0 then
first_line = atom_body_token_source_line(atom, token, line_in_body)
end
end
end
if touched then
findings[#findings + 1] = {
atom = atom.name or "",
line = first_line,
check = "gte_cr_TR_naming",
kind = "info",
msg = string.format(
"atom '%s' uses gte_cr_TR[XYZ]; translation-vector registers are the only "
.. "3-letter-suffix C2 aliases (TRX/TRY/TRZ). See docs/gte_reference.md §"
.. "\"The `gte_cmdw_mvmva_c11_pass2_exact` literal\" for the libgte outer-product "
.. "convention that uses these names."
, atom.name or ""),
}
end
end
-- check_immediate_field_width — flags integer literals passed to instruction
-- macros that exceed the immediate field width. Reads `IMMEDIATE_FIELD_WIDTHS`
-- from duffle.lua. Only fires on parseable integer literals; register names,
-- O_(...) offsets, atom_offset(...) markers, and enum tokens are skipped.
local function check_immediate_field_width(atom, pipe_ctx, findings)
local widths = duffle.IMMEDIATE_FIELD_WIDTHS or {}
local events = atom.paths and atom.paths.word_events or {}
local line_for_word_event = pipe_ctx.line_for_word_event
for _, ev in ipairs(events) do
local ev_ident = ev.encoder or ev.ident or "?"
local rules = widths[ev_ident]
if rules then
local ev_args = ev.args or {}
local ev_line = line_for_word_event and line_for_word_event(ev) or atom.line
for _, rule in ipairs(rules) do
local arg_str = ev_args[rule.arg]
if arg_str then
local value = parse_integer_literal(arg_str)
if value then
local width = rule.width
local is_signed = rule.signed == true
-- parse_integer_literal returns a U4-wrapped value in [0, 2^32).
-- For signed fields, re-interpret the high bit as the sign.
local signed_value = value
if is_signed and value >= 0x80000000 then
signed_value = value - 0x100000000
end
local lo, hi
if is_signed then
lo = -(bit.lshift(1, width - 1))
hi = bit.lshift(1, width - 1) - 1
else
lo = 0
hi = bit.lshift(1, width) - 1
end
-- For unsigned fields, a negative C literal (high bit set in U4)
-- is valid if the low `width` bits fit — IMM_MASK truncates it.
-- Flag as a warning (code smell), not an error.
local check_value = is_signed and signed_value or value
local field_max = bit.lshift(1, width) - 1
local low_bits_fit = (value % (bit.lshift(1, width))) == value or (is_signed and signed_value >= lo and signed_value <= hi)
if is_signed then
if signed_value < lo or signed_value > hi then
findings[#findings + 1] = {
check = "immediate_field_width",
kind = "error",
atom = atom.name,
line = ev_line,
msg = string.format(
"%s: immediate %d at arg %d overflows %d-bit %s field (valid %d..%d)",
ev_ident, signed_value, rule.arg, width,
"signed", lo, hi),
}
end
else
-- Unsigned field: check if the low `width` bits exceed the field.
-- A negative C literal (U4 >= 0x80000000) whose low bits fit is
-- valid but a code smell — warn, don't error.
local low_bits = value % (bit.lshift(1, width))
if value > field_max then
if value >= 0x80000000 and low_bits <= field_max then
findings[#findings + 1] = {
check = "immediate_field_width",
kind = "warning",
atom = atom.name,
line = ev_line,
msg = string.format(
"%s: negative immediate %d at arg %d on unsigned %d-bit field (truncated to %d by IMM_MASK)",
ev_ident, signed_value, rule.arg, width, low_bits),
}
else
findings[#findings + 1] = {
check = "immediate_field_width",
kind = "error",
atom = atom.name,
line = ev_line,
msg = string.format(
"%s: immediate %d at arg %d overflows %d-bit unsigned field (valid 0..%d)",
ev_ident, value, rule.arg, width, field_max),
}
end
end
end
end
end
end
end
end
end
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
-- ════════════════════════════════════════════════════════════════════════════
@@ -2414,6 +2682,10 @@ local CHECK_RULES = {
{ name = "abi_handoff", per_atom = check_abi_handoff },
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
{ name = "gte_cr_alias_writes", per_atom = check_gte_cr_alias_writes },
{ name = "rtdiagonal_completeness", per_atom = check_rtdiagonal_completeness },
{ name = "gte_cr_TR_naming", per_atom = check_gte_cr_TR_naming },
{ name = "immediate_field_width", per_atom = check_immediate_field_width },
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
@@ -2457,6 +2729,11 @@ local function build_corpus_pipe_ctx(ctx)
atom_infos_list = corpus.atom_infos or {},
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
collisions = corpus.collisions or {},
-- GTE control-register alias groups (from `duffle.GTE_CR_ALIAS_GROUPS`).
-- The three new per_atom checks (gte_cr_alias_writes, rtdiagonal_completeness,
-- gte_cr_TR_naming) read from this view. `duffle` is exposed alongside so
-- `find_alias_pair_for` can resolve alias → group without a separate registry.
gte_cr_alias_groups = duffle.GTE_CR_ALIAS_GROUPS or {},
}
end
Binary file not shown.
+124
View File
@@ -61,6 +61,119 @@ if (-not $msbuild_exe) {
}
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
# ════════════════════════════════════════════════════════════════════════════
# NuGet restore — required before MSBuild.
# pcsx-redux's .vcxproj files use the legacy packages.config style with
# hardcoded `<Import Project="..\packages\{id}.{ver}\...">` directives.
# MSBuild's `/t:Restore` won't fetch missing packages here (the local
# packages\ dir is checked but no package-source lookup happens), and
# `dotnet restore` errors on packages.config projects, so we walk every
# packages.config, parse out the <package id version/> entries, and pull
# any missing .nupkg directly from api.nuget.org's flat container.
# ════════════════════════════════════════════════════════════════════════════
$path_pcsx_packages = join-path $path_pcsx_redux 'vsprojects\packages'
$nuget_flat_container = 'https://api.nuget.org/v3-flatcontainer'
# Collect required (id, version) pairs from every packages.config.
$required_packages = @{}
Get-ChildItem -Path (join-path $path_pcsx_redux 'vsprojects') -Filter 'packages.config' -Recurse -ErrorAction SilentlyContinue |
ForEach-Object {
[xml]$xml = Get-Content -LiteralPath $_.FullName -Raw
foreach ($pkg in $xml.packages.package) {
$key = '{0}|{1}' -f $pkg.id, $pkg.version
$required_packages[$key] = @{ id = $pkg.id; version = $pkg.version }
}
}
# Ensure the packages root exists.
if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
}
# Download anything missing. Skip the package entirely if its dir already has
# any contents (the legacy packages.config style means the targets file
# location varies per package — `luajit.native` puts it at build/native/,
# `glfw` puts it elsewhere — so we can't probe a specific path; just check
# whether the dir is non-empty).
Add-Type -AssemblyName System.IO.Compression.FileSystem
foreach ($pkg in $required_packages.Values) {
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
if ((Test-Path -LiteralPath $pkgDir) -and `
(@(Get-ChildItem -LiteralPath $pkgDir -Recurse -ErrorAction SilentlyContinue).Count -gt 0)) {
continue
}
$url = '{0}/{1}/{2}/{1}.{2}.nupkg' -f $nuget_flat_container, $pkg.id, $pkg.version
$nupkg = Join-Path $pkgDir ('{0}.{1}.nupkg' -f $pkg.id, $pkg.version)
New-Item -ItemType Directory -Path $pkgDir -Force | Out-Null
Write-Host "Fetching NuGet package: $($pkg.id) $($pkg.version)"
try {
Invoke-WebRequest -Uri $url -OutFile $nupkg -UseBasicParsing -ErrorAction Stop
[System.IO.Compression.ZipFile]::ExtractToDirectory($nupkg, $pkgDir)
Remove-Item -LiteralPath $nupkg -Force
} catch {
$msg = $_.Exception.Message
if ($msg -match '404') {
Write-Host " Not on nuget.org (vendored?) — skipping $url"
} else {
Write-Warning "Failed to fetch $url$msg"
}
if (Test-Path -LiteralPath $nupkg) { Remove-Item -LiteralPath $nupkg -Force }
}
}
# ════════════════════════════════════════════════════════════════════════════
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into
# luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick. The raw string
# literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or
# MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384,
# minus 5 bytes for the `-- lualoader, ` prefix). If the upstream file
# grows past that, trim it: remove license header, trailing whitespace,
# blank separators, inline comments, and shrink 4-space indent to 2-space.
# Idempotent — only writes when the raw string exceeds the limit.
# ════════════════════════════════════════════════════════════════════════════
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
if (Test-Path -LiteralPath $path_isoffi) {
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
$startMarker = $content.IndexOf('R"EOF(--')
$endMarker = $content.IndexOf('-- )EOF"')
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) {
$endMarker - ($startMarker + 8)
} else { -1 }
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
if ($literalLen -gt 16379) {
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
$lines = $content -split "`n"
$markerIdx = -1
for ($i = 0; $i -lt $lines.Length; $i++) {
if ($lines[$i] -match '^-- \)EOF"') { $markerIdx = $i; break }
}
$newLines = @()
for ($i = 0; $i -lt $lines.Length; $i++) {
$lineNum = $i + 1
$line = $lines[$i]
# Keep the first line and the EOF-marker line untouched.
if ($i -eq 0 -or $i -eq $markerIdx) { $newLines += $line; continue }
# Drop the GPL license header (lines 2-17).
if ($lineNum -ge 2 -and $lineNum -le 17) { continue }
# Drop blank separator lines.
if ($line -match '^\s*$') { continue }
# Drop trailing whitespace.
$line = $line -replace '\s+$', ''
# Drop inline comments (anything from `--` to end of line).
$line = $line -replace '\s*--.*$', ''
# Shrink 4-space indent to 2-space.
$line = $line -replace '^( )', ' '
if ($line -match '^\s*$') { continue }
$newLines += $line
}
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) `
-replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
}
}
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
@@ -117,6 +230,17 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
# ════════════════════════════════════════════════════════════════════════════
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
# Wipe stale *.dep files across src\mips. These cache absolute paths to the
# GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
# `make clean` in openbios only clears its own dir — subdirs like
# common/crt0/, modplayer/, and shell/ keep their stale .dep files. Easier to
# just delete the lot before each build than to teach every Makefile about
# deepclean recursion.
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
push-location $path_openbios
& make clean
& make