mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-08-15 03:58:15 +00:00
Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
004a7eff19 | ||
|
|
e42c75a26a | ||
|
|
69f2c0d036 | ||
|
|
b045856dd6 | ||
|
|
68b87f1c8b | ||
|
|
917b764d95 | ||
|
|
773aa44013 | ||
|
|
2b6fe53ce8 |
@@ -20,3 +20,4 @@ toolchain/lpeg
|
||||
|
||||
scratch
|
||||
toolchain/libpsn00b
|
||||
scripts/pcsx_debug_helper.zip
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
#endif
|
||||
enum {
|
||||
bios_init_pad_2 = 0x12,
|
||||
bios_start_pad_2 = 0x13,
|
||||
bios_flushcache = 0x44,
|
||||
bios_table_addr = 0xA0,
|
||||
bios_btable_addr = 0xB0,
|
||||
};
|
||||
|
||||
enum {
|
||||
bios_pad_buffer_size = 0x22,
|
||||
};
|
||||
@@ -75,6 +75,26 @@
|
||||
* ----------------------------------------------------------------------------*/
|
||||
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
|
||||
// enum {
|
||||
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
|
||||
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
|
||||
// };
|
||||
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
|
||||
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
|
||||
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
|
||||
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
|
||||
// enum {
|
||||
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
|
||||
// phase_auto_reg(cube_g4, R_Temp1),
|
||||
// };
|
||||
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
|
||||
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
|
||||
|
||||
/* ============================================================================
|
||||
* atom_info :
|
||||
* MipsAtom_(cube_tri) atom_info(
|
||||
|
||||
+7
-3
@@ -28,8 +28,9 @@
|
||||
#define internal static // internal
|
||||
|
||||
#define asm __asm__
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
|
||||
#define A_(data) (& data)
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||
@@ -133,8 +134,8 @@ typedef __UINT32_TYPE__ TSet_(B4);
|
||||
#define u4_v(value) C_(U4 V_*, value)
|
||||
enum { false = 0, true = 1, true_overflow, };
|
||||
|
||||
#define u4_lo(value) ((value) & 0xFFFFU)
|
||||
#define u4_hi(value) ((value) >> 12)
|
||||
#define u4_lo(value) (u4_(value) & 0xFFFFU)
|
||||
#define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
|
||||
|
||||
typedef void Proc_(VoidFn) (void);
|
||||
|
||||
@@ -168,6 +169,8 @@ def_signed_ops(le, <=)
|
||||
#undef def_signed_ops
|
||||
#undef def_signed_op
|
||||
|
||||
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
|
||||
#if 0
|
||||
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
||||
#define add_s(a,b) def_generic_sop(add,a,b)
|
||||
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
||||
@@ -177,6 +180,7 @@ def_signed_ops(le, <=)
|
||||
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
||||
#define le_s(a,b) def_generic_sop(le, a,b)
|
||||
#undef def_generic_sop
|
||||
#endif
|
||||
|
||||
#define alignas _Alignas
|
||||
#define alignof _Alignof
|
||||
|
||||
+140
-15
@@ -14,7 +14,9 @@
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||
@@ -68,6 +70,27 @@ WORD_COUNT(mac_load_v2s2, 2)
|
||||
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
||||
WORD_COUNT(mac_store_v2s2, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
|
||||
load_word( rs_x, r_base, O_(V3_S4,x)) \
|
||||
, load_word( rs_y, r_base, O_(V3_S4,y)) \
|
||||
, load_word( rs_z, r_base, O_(V3_S4,z))
|
||||
WORD_COUNT(mac_load_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
||||
store_word(rt_x, base, offset + O_(V3_S4,x)) \
|
||||
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
|
||||
, store_word(rt_z, base, offset + O_(V3_S4,z))
|
||||
WORD_COUNT(mac_store_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
|
||||
sub_s(rds_x, rds_x, rt_x) \
|
||||
, sub_s(rds_y, rds_y, rt_y) \
|
||||
, sub_s(rds_z, rds_z, rt_z)
|
||||
WORD_COUNT(mac_sub_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||
@@ -124,6 +147,96 @@ WORD_COUNT(mac_gte_store_g4_p012, 3)
|
||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
|
||||
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
|
||||
gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||
, nop \
|
||||
, gte_cmdw_sqr \
|
||||
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
|
||||
WORD_COUNT(mac_gte_sqr_v3, 8)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_dz, C2_MAC3) \
|
||||
, shift_aright_var(r_dx, r_dx, r_shift) \
|
||||
, shift_aright_var(r_dy, r_dy, r_shift) \
|
||||
, shift_aright_var(r_dz, r_dz, r_shift)
|
||||
WORD_COUNT(mac_gte_gpf_scale, 13)
|
||||
|
||||
#define mac_normalize_v3s4(...) \
|
||||
load_word(r_src, R_TapePtr, O_(Binds_NormalizeV3S4,src)) /* pop src ptr (scratch addr) */ \
|
||||
, load_word(r_dst, R_TapePtr, O_(Binds_NormalizeV3S4,dst)) /* pop dst ptr (scratch addr) */ \
|
||||
, add_ui_self( R_TapePtr, S_(Binds_NormalizeV3S4)) \
|
||||
, load_word(r_sx, r_src, O_(V3_S4,x)) \
|
||||
, load_word(r_sy, r_src, O_(V3_S4,y)) \
|
||||
, load_word(r_sz, r_src, O_(V3_S4,z)) \
|
||||
, nop /* load-delay */ /* ── 48-word normalize body (preserved verbatim from ac_normalize_v3s4) ─────── */ /* Stage 1: mtc2 src → IR1/2/3, SQR fires (MAC1/2/3 = IR², IR ← MAC saturated) */ \
|
||||
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||
, nop \
|
||||
, gte_cmdw_sqr /* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS */ \
|
||||
, gte_mv_from_data_r(r_sq_y, C2_MAC1) /* r_sq_y = MAC1 = sx² */ \
|
||||
, gte_mv_from_data_r(r_sq_z, C2_MAC2) /* r_sq_z = MAC2 = sy² */ \
|
||||
, gte_mv_from_data_r(r_recip_est, C2_MAC3) /* r_recip_est = MAC3 = sz² */ \
|
||||
, nop /* MFC2→GPR load delay (1 slot) */ \
|
||||
, add_u(r_recip_est, r_recip_est, r_sq_z) /* r_recip_est += sy² */ \
|
||||
, add_u(r_recip_est, r_recip_est, r_sq_y) /* r_recip_est += sx² (sum = |v|²) */ \
|
||||
, gte_mv_to_data_r( r_recip_est, C2_LZCS) /* LZCS = |v|² */ \
|
||||
, nop2 \
|
||||
, gte_mv_from_data_r(r_lzcr, C2_LZCR) /* r_lzcr = LZCR (count of leading bits) */ \
|
||||
, nop /* MFC2→GPR load delay (1 slot) */ /* Stage 3: compute shift amount, align |v|² to bit 24, lookup 1/|v| */ \
|
||||
, and_i( r_lzcr, r_lzcr, -2) /* r_lzcr &= ~1 (force even for halving) */ \
|
||||
, li_s( r_shift, 31) /* r_shift = 31 */ \
|
||||
, sub_s( r_shift, r_shift, r_lzcr) /* r_shift = 31 - LZCR */ \
|
||||
, shift_aright( r_shift, r_shift, 1) /* r_shift = (31 - LZCR) / 2 */ \
|
||||
, add_si( r_tmp, r_lzcr, -24) /* r_tmp = LZCR - 24 (signed, for branch) */ \
|
||||
, branch_lt_zero(r_tmp, atom_offset(srav_path, aligned_done)) \
|
||||
, nop \
|
||||
, jump_rel( atom_offset(aligned_done, srav_path)) \
|
||||
, shift_lleft_var(r_recip_est, r_recip_est, r_tmp) /* BD-slot of branch_equal: r_recip_est = |v|² << (LZCR - 24) */ \
|
||||
, atom_label(srav_path) /* SRAV path: |v|² is small (top bit < bit 24) */ \
|
||||
, li_s( r_tmp, 24) \
|
||||
, sub_s( r_tmp, r_tmp, r_lzcr) /* r_tmp = 24 - LZCR */ \
|
||||
, shift_aright_var(r_recip_est, r_recip_est, r_tmp) /* r_recip_est = |v|² >> (24 - LZCR) */ \
|
||||
, atom_label(aligned_done) /* Both paths converge here with |v|² aligned to bit 24 */ /* r_recip_est now holds |v|² aligned to bit 24 — convert to byte offset, -64 to skip zero pad. */ \
|
||||
, add_si( r_recip_est, r_recip_est, -64) \
|
||||
, shift_lleft( r_recip_est, r_recip_est, 1) /* r_recip_est *= 2 (half-word index) */ /* Reference OUR local sqrtbl via &-address split. Compiler/linker resolves both halves. */ \
|
||||
, load_upper_i( r_tmp, u4_hi(& gte_normalize_sqr_tbl)) /* lui */ \
|
||||
, or_i_self( r_tmp, u4_lo(& gte_normalize_sqr_tbl)) /* ori */ \
|
||||
, add_u( r_tmp, r_tmp, r_recip_est) /* r_tmp = sqrtbl base + byte offset (matches libgte 0x80016118: addu t5,t5,t4) */ \
|
||||
, load_half( r_recip_est, r_tmp, 0) /* r_recip_est = sqrtbl[r_recip_est] = 1/|v| estimate */ \
|
||||
, nop /* retire load_half before MTC2 (matches libgte 0x80016120: nop) */ /* Stage 4: mtc2 IR0..3, GPF (MAC = IR0*IR), mfc2 MAC, srav finalize */ \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_IR0) /* IR0 = 1/|v| estimate */ \
|
||||
, gte_mv_to_data_r(r_sx, C2_IR1) /* IR1 = src.x */ \
|
||||
, gte_mv_to_data_r(r_sy, C2_IR2) /* IR2 = src.y */ \
|
||||
, gte_mv_to_data_r(r_sz, C2_IR3) /* IR3 = src.z */ \
|
||||
, nop2 /* COP2 transfer latency (2 slots) */ \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(r_sx, C2_MAC1) /* MAC1 → r_sx (overwrites src.x with raw reciprocal-scaled) */ \
|
||||
, gte_mv_from_data_r(r_sy, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_sz, C2_MAC3) \
|
||||
, shift_aright_var(r_sx, r_sx, r_shift) \
|
||||
, shift_aright_var(r_sy, r_sy, r_shift) \
|
||||
, shift_aright_var(r_sz, r_sz, r_shift) /* ── I/O wrapper tail (~3 words) ───────────────────────────────────────────── */ \
|
||||
, store_word(r_sx, r_dst, O_(V3_S4,x)) \
|
||||
, store_word(r_sy, r_dst, O_(V3_S4,y)) \
|
||||
, store_word(r_sz, r_dst, O_(V3_S4,z)) /* ── atom_reads(R_TapePtr) atom_writes(R_TapePtr) ────────────────────────── */ \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_normalize_v3s4, 62)
|
||||
|
||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||
load_upper_i(reg_transfer, cmd >> 16) \
|
||||
, or_i_self( reg_transfer, cmd & 0xFFFF) \
|
||||
@@ -156,29 +269,41 @@ WORD_COUNT(mac_format_f3_color, 3)
|
||||
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||
WORD_COUNT(mac_format_g4_color, 12)
|
||||
|
||||
#define mac_insert_ot_tag_f3(r_ot_base, r_prim_cursor) \
|
||||
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
|
||||
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
|
||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag_f3, 11)
|
||||
WORD_COUNT(mac_insert_ot_tag, 11)
|
||||
|
||||
#define mac_insert_ot_tag_g4(r_ot_base, r_prim_cursor) \
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
|
||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||
WORD_COUNT(mac_insert_ot_tag_g4, 11)
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_set_centered_axes(r_state, r_scratch) \
|
||||
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF) \
|
||||
, or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF) \
|
||||
, store_word( r_scratch, r_state, O_(PadState,axes))
|
||||
WORD_COUNT(mac_pad_set_centered_axes, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_set_id_byte(r_state, r_id, id_value) \
|
||||
add_ui( r_id, R_0, id_value) \
|
||||
, store_byte(r_id, r_state, O_(PadState,id))
|
||||
WORD_COUNT(mac_pad_set_id_byte, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
|
||||
add_ui( r_tmp, R_0, pad_status) \
|
||||
, store_word(r_tmp, r_state, O_(PadState,status))
|
||||
WORD_COUNT(mac_pad_set_status, 2)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
|
||||
nor_u( r_buttons, r_buttons, R_0) \
|
||||
, store_half( r_buttons, r_pad_state, O_(PadState, buttons))
|
||||
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
|
||||
|
||||
|
||||
+22
-10
@@ -11,7 +11,9 @@
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||
@@ -23,17 +25,27 @@
|
||||
#pragma region duffle
|
||||
|
||||
|
||||
// --- atom: pad_bios_snapshot (78 words) ---
|
||||
// --- atom: normalize_v3s4 (62 words) ---
|
||||
|
||||
#define _atom_offset_snap_root_skip_disconnected 8
|
||||
#define _atom_offset_disconnected_snap_end 61
|
||||
#define _atom_offset_case_2_id_dispatch 8
|
||||
#define _atom_offset_pending_snap_end 51
|
||||
#define _atom_offset_id_dispatch_try_analog_stick 11
|
||||
#define _atom_offset_id_dispatch_snap_end 38
|
||||
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
||||
#define _atom_offset_analog_stick_snap_end 24
|
||||
#define _atom_offset_try_analog_pad_try_unsupported 11
|
||||
#define _atom_offset_srav_path_aligned_done 6
|
||||
#define _atom_offset_aligned_done_srav_path 1
|
||||
|
||||
enum {
|
||||
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
||||
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
||||
};
|
||||
|
||||
// --- atom: pad_bios_snapshot (84 words) ---
|
||||
|
||||
#define _atom_offset_snap_root_skip_disconnected 10
|
||||
#define _atom_offset_disconnected_snap_end 65
|
||||
#define _atom_offset_case_2_id_dispatch 9
|
||||
#define _atom_offset_pending_snap_end 54
|
||||
#define _atom_offset_id_dispatch_try_analog_stick 12
|
||||
#define _atom_offset_id_dispatch_snap_end 40
|
||||
#define _atom_offset_try_analog_stick_try_analog_pad 13
|
||||
#define _atom_offset_analog_stick_snap_end 25
|
||||
#define _atom_offset_try_analog_pad_try_unsupported 12
|
||||
#define _atom_offset_analog_pad_snap_end 10
|
||||
|
||||
enum {
|
||||
|
||||
+12
-34
@@ -8,54 +8,47 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
FI_ Slice_MipsCode ac_gcmd_push(U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_gcmd_push, {
|
||||
FI_ Slice_MipsCode ac_gcmd_push(MipsAtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_gcmd_push, ab, {
|
||||
load_upper_i(reg_transfer, cmd >> 16),
|
||||
or_i_self( reg_transfer, cmd & 0xFFFF),
|
||||
store_word( reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, {
|
||||
FI_ Slice_MipsCode ac_store_rgb8(MipsAtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, {
|
||||
store_byte(rr, base, offset + O_(RGB8,r)),
|
||||
store_byte(rg, base, offset + O_(RGB8,g)),
|
||||
store_byte(rb, base, offset + O_(RGB8,b)),
|
||||
})
|
||||
|
||||
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
|
||||
* byte offset. Internal helper used by the *_format_*_color macros. */
|
||||
FI_ Slice_MipsCode ac_pack_color_word(U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
|
||||
FI_ Slice_MipsCode ac_pack_color_word(MipsAtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, {
|
||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||
store_word( R_AT, r_base, (off)),
|
||||
})
|
||||
|
||||
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
||||
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
|
||||
FI_ Slice_MipsCode ac_format_f3_color(U4 r_base, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||
FI_ Slice_MipsCode ac_format_f3_color(MipsAtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
||||
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||
|
||||
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
|
||||
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
|
||||
FI_ Slice_MipsCode ac_format_g4_color(U4 r_prim_cursor,
|
||||
FI_ Slice_MipsCode ac_format_g4_color(MipsAtomBuilder_R ab, U4 r_prim_cursor,
|
||||
U1 r0, U1 g0, U1 b0,
|
||||
U1 r1, U1 g1, U1 b1,
|
||||
U1 r2, U1 g2, U1 b2,
|
||||
U1 r3, U1 g3, U1 b3)
|
||||
MipsAtomComp_Proc_(ac_format_g4_color, {
|
||||
MipsAtomComp_Proc_(ac_format_g4_color, ab, {
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||
})
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
||||
I_ Slice_MipsCode ac_insert_ot_tag_f3(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_f3, {
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
||||
I_ Slice_MipsCode ac_insert_ot_tag(MipsAtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
|
||||
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||
@@ -64,19 +57,4 @@ I_ Slice_MipsCode ac_insert_ot_tag_f3(U4 r_ot_base, U4 r_prim_cursor) MipsAtomCo
|
||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||
})
|
||||
|
||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
||||
I_ Slice_MipsCode ac_insert_ot_tag_g4(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_g4, {
|
||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
|
||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||
})
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
+233
-11
@@ -11,7 +11,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||
FI_ Slice_MipsCode ac_load_tri_indices(U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, {
|
||||
FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, {
|
||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
||||
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
||||
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||
@@ -19,14 +19,14 @@ FI_ Slice_MipsCode ac_load_tri_indices(U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i
|
||||
|
||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||
FI_ Slice_MipsCode ac_gte_store_f3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, {
|
||||
FI_ Slice_MipsCode ac_gte_store_f3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, {
|
||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
||||
})
|
||||
|
||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||
I_ Slice_MipsCode ac_gte_load_tri_verts(U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, {
|
||||
I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, {
|
||||
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||
@@ -37,7 +37,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v
|
||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||
FI_ Slice_MipsCode ac_gte_store_g4_p012(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, {
|
||||
FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, {
|
||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||
@@ -47,22 +47,244 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(U4 r_primitive_cursor) atom_dbg_skip Mip
|
||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||
*/
|
||||
FI_ Slice_MipsCode ac_gte_store_g4_p3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||
FI_ Slice_MipsCode ac_gte_store_g4_p3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||
|
||||
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
||||
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
|
||||
* Stage 2 of normalize consumes these directly.
|
||||
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
|
||||
FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, {
|
||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||
nop, gte_cmdw_sqr,
|
||||
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
||||
gte_mv_from_data_r(r_sq_y, C2_MAC2),
|
||||
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
||||
})
|
||||
|
||||
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
||||
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
|
||||
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
||||
* Used standalone for "scale vector by scalar".
|
||||
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
||||
FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, {
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(r_dx, C2_MAC1),
|
||||
gte_mv_from_data_r(r_dy, C2_MAC2),
|
||||
gte_mv_from_data_r(r_dz, C2_MAC3),
|
||||
shift_aright_var(r_dx, r_dx, r_shift),
|
||||
shift_aright_var(r_dy, r_dy, r_shift),
|
||||
shift_aright_var(r_dz, r_dz, r_shift),
|
||||
})
|
||||
|
||||
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
|
||||
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
||||
*
|
||||
* Data is identical to the libgte original (byte-for-byte verified).
|
||||
*
|
||||
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
|
||||
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
|
||||
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
|
||||
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
|
||||
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
|
||||
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
|
||||
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
|
||||
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
|
||||
* Sampling the first value of each octave:
|
||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
|
||||
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
|
||||
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
|
||||
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
|
||||
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
|
||||
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
|
||||
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
|
||||
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
|
||||
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
|
||||
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
|
||||
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
|
||||
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
|
||||
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
|
||||
*
|
||||
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
|
||||
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
|
||||
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
|
||||
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
|
||||
* and the load upper_halves of the table bracket the input range.
|
||||
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||
*
|
||||
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
|
||||
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
||||
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
|
||||
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
|
||||
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
|
||||
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
|
||||
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
|
||||
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
|
||||
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
|
||||
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
|
||||
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
|
||||
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
|
||||
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
|
||||
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
|
||||
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
|
||||
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
|
||||
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
|
||||
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
|
||||
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
|
||||
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
|
||||
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
|
||||
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
|
||||
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||
};
|
||||
|
||||
/* ─── Full normalize (all 4 stages inline) ───
|
||||
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
|
||||
*
|
||||
* Component variants that could apply:
|
||||
* - `ac_gte_sqr_v3` (line ~56) covers stage 1's `mtc2 IR1/2/3 + nop + gte_cmdw_sqr`.
|
||||
* We do NOT call it because the inlined version of stage 1 is followed immediately by stage 2's `mfc2 MAC1/2/3` chain
|
||||
* (the operands of `ac_gte_sqr_v3`'s r_sq_x/r_sq_y/r_sq_z would each require an explicit GPR to receive the MAC result,
|
||||
* then a move to land in r_recip_est for the partial-sum chain).
|
||||
* Inlining saves ~3 cycles of `or`-merge + register pressure
|
||||
* (squared MAC3 lands DIRECTLY in r_recip_est which doubles as the partial-sum accumulator and the LZCS input — see r_recip_est row below).
|
||||
* - `ac_gte_gpf_scale` (line ~71) covers stage 4's `mtc2 IR0..3 + nop2 + gte_cmdw_gpf + mfc2 MAC1/2/3 + sra`.
|
||||
* We do NOT call it for the symmetric reason: the normalize in-place semantics overwrite the input regs (r_sx/r_sy/r_sz) with the normalized output,
|
||||
* which `ac_gte_gpf_scale`'s r_dx/r_dy/r_dz output GPRs would not match.
|
||||
* `gte_cmdw_sqr` and `gte_cmdw_gpf` primitive macros ARE used in the inlined body, so changes to those primitives
|
||||
* (e.g., the libgte `fake_cmd` signature bits) propagate automatically. The components remain available for callers that want the explicit GPR-shape variants.
|
||||
*
|
||||
* Argument aliasing (9 unique physical regs needed, can drop to 8 with r_sq_y ≡ r_lzcr):
|
||||
* r_sx, r_sy, r_sz : src components in regs (clobbered by mtc2 → IR1/2/3 in stage 1, then by mfc2 MAC1/2/3 in stage 4 — in-place semantics)
|
||||
* r_sq_y, r_sq_z : MAC2, MAC3 → DIE after stage 2 accumulate (r_sq_y can alias r_lzcr after stage 2 to save one reg)
|
||||
* r_recip_est : ≡ r_sqmag — multi-purpose (holds |v|² in stage 2, shift-input in stage 3, sqrtbl[index] in stage 4)
|
||||
* r_lzcr : LZCR value, alive across stage 3 (srav path needs `24 - LZCR`)
|
||||
* r_shift : (31 - LZCR & ~1) >> 1 — final srav amount (stages 3-4)
|
||||
* r_tmp : scratch (shift count, branch target, lookup addr, table base)
|
||||
*
|
||||
* GPR ccount peak: 9.
|
||||
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
|
||||
* Words: ~35 (pending re-gen; matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
|
||||
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local. */
|
||||
|
||||
/* ─── Binds_NormalizeV3S4 — declared here so the MipsAtom_Proc_ body can reference
|
||||
* O_(Binds_NormalizeV3S4,*). Inlined at the proc-call site; not exposed in gen/macs.h. */
|
||||
typedef Struct_(Binds_NormalizeV3S4) {
|
||||
U4 src; /* V3_S4* (scratch address — read from tape) */
|
||||
U4 dst; /* V3_S4* (scratch address — write to tape) */
|
||||
};
|
||||
|
||||
/* NOTE: The bundle-specific scratchpad offset schema was intentionally kept out of this file.
|
||||
* gte.atom.c is the GENERIC GTE primitives file — it exposes only the parameter-style normalize_v3s4_proc for any future caller. */
|
||||
I_ void normalize_v3s4_proc(
|
||||
MipsAtomBuilder_R ab
|
||||
, U4 r_src /* GPR code: scratch base carrier (wave-context, e.g., R_T4) */
|
||||
, U4 r_dst /* GPR code: scratch dst pointer carrier (wave-context, e.g., R_T5) */
|
||||
, U4 r_sx, U4 r_sy, U4 r_sz /* GPR codes: src.x/y/z scratch (atom-local) */
|
||||
, U4 r_sq_y, U4 r_sq_z /* GPR codes: MAC1/2 scratch (atom-local) */
|
||||
, U4 r_recip_est /* GPR code: |v|² sum + shift-input + sqrtbl[index] (atom-local) */
|
||||
, U4 r_lzcr /* GPR code: LZCR value (atom-local) */
|
||||
, U4 r_shift /* GPR code: final srav amount (atom-local) */
|
||||
, U4 r_tmp /* GPR code: scratch (shift count, branch target, lookup addr, table base) */
|
||||
)
|
||||
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
|
||||
MipsAtom_Proc_(normalize_v3s4, ab, {
|
||||
/* ── I/O wrapper (~10 words: 3 bind-pop + 3 src-load + 1 nop + 3 dst-store) ─── */
|
||||
load_word(r_src, R_TapePtr, O_(Binds_NormalizeV3S4,src)), /* pop src ptr (scratch addr) */
|
||||
load_word(r_dst, R_TapePtr, O_(Binds_NormalizeV3S4,dst)), /* pop dst ptr (scratch addr) */
|
||||
add_ui_self( R_TapePtr, S_(Binds_NormalizeV3S4)),
|
||||
load_word(r_sx, r_src, O_(V3_S4,x)),
|
||||
load_word(r_sy, r_src, O_(V3_S4,y)),
|
||||
load_word(r_sz, r_src, O_(V3_S4,z)),
|
||||
nop, /* load-delay */
|
||||
|
||||
/* ── 48-word normalize body (preserved verbatim from ac_normalize_v3s4) ─────── */
|
||||
// Stage 1: mtc2 src → IR1/2/3, SQR fires (MAC1/2/3 = IR², IR ← MAC saturated)
|
||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||
nop, gte_cmdw_sqr,
|
||||
// Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS
|
||||
gte_mv_from_data_r(r_sq_y, C2_MAC1), /* r_sq_y = MAC1 = sx² */
|
||||
gte_mv_from_data_r(r_sq_z, C2_MAC2), /* r_sq_z = MAC2 = sy² */
|
||||
gte_mv_from_data_r(r_recip_est, C2_MAC3), /* r_recip_est = MAC3 = sz² */
|
||||
nop, /* MFC2→GPR load delay (1 slot) */
|
||||
add_u(r_recip_est, r_recip_est, r_sq_z), /* r_recip_est += sy² */
|
||||
add_u(r_recip_est, r_recip_est, r_sq_y), /* r_recip_est += sx² (sum = |v|²) */
|
||||
gte_mv_to_data_r( r_recip_est, C2_LZCS), /* LZCS = |v|² */
|
||||
nop2,
|
||||
gte_mv_from_data_r(r_lzcr, C2_LZCR), /* r_lzcr = LZCR (count of leading bits) */
|
||||
nop, /* MFC2→GPR load delay (1 slot) */
|
||||
// Stage 3: compute shift amount, align |v|² to bit 24, lookup 1/|v|
|
||||
and_i( r_lzcr, r_lzcr, -2), /* r_lzcr &= ~1 (force even for halving) */
|
||||
li_s( r_shift, 31), /* r_shift = 31 */
|
||||
sub_s( r_shift, r_shift, r_lzcr), /* r_shift = 31 - LZCR */
|
||||
shift_aright( r_shift, r_shift, 1), /* r_shift = (31 - LZCR) / 2 */
|
||||
add_si( r_tmp, r_lzcr, -24), /* r_tmp = LZCR - 24 (signed, for branch) */
|
||||
branch_lt_zero(r_tmp, atom_offset(srav_path, aligned_done)), nop,
|
||||
jump_rel( atom_offset(aligned_done, srav_path)),
|
||||
shift_lleft_var(r_recip_est, r_recip_est, r_tmp), /* BD-slot of branch_equal: r_recip_est = |v|² << (LZCR - 24) */
|
||||
atom_label(srav_path) /* SRAV path: |v|² is small (top bit < bit 24) */
|
||||
li_s( r_tmp, 24),
|
||||
sub_s( r_tmp, r_tmp, r_lzcr), /* r_tmp = 24 - LZCR */
|
||||
shift_aright_var(r_recip_est, r_recip_est, r_tmp), /* r_recip_est = |v|² >> (24 - LZCR) */
|
||||
atom_label(aligned_done) /* Both paths converge here with |v|² aligned to bit 24 */
|
||||
/* r_recip_est now holds |v|² aligned to bit 24 — convert to byte offset, -64 to skip zero pad. */
|
||||
add_si( r_recip_est, r_recip_est, -64),
|
||||
shift_lleft( r_recip_est, r_recip_est, 1), /* r_recip_est *= 2 (half-word index) */
|
||||
/* Reference OUR local sqrtbl via &-address split. Compiler/linker resolves both halves. */
|
||||
load_upper_i( r_tmp, u4_hi(& gte_normalize_sqr_tbl)), /* lui */
|
||||
or_i_self( r_tmp, u4_lo(& gte_normalize_sqr_tbl)), /* ori */
|
||||
add_u( r_tmp, r_tmp, r_recip_est), /* r_tmp = sqrtbl base + byte offset (matches libgte 0x80016118: addu t5,t5,t4) */
|
||||
load_half( r_recip_est, r_tmp, 0), /* r_recip_est = sqrtbl[r_recip_est] = 1/|v| estimate */
|
||||
nop, /* retire load_half before MTC2 (matches libgte 0x80016120: nop) */
|
||||
// Stage 4: mtc2 IR0..3, GPF (MAC = IR0*IR), mfc2 MAC, srav finalize
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0), /* IR0 = 1/|v| estimate */
|
||||
gte_mv_to_data_r(r_sx, C2_IR1), /* IR1 = src.x */
|
||||
gte_mv_to_data_r(r_sy, C2_IR2), /* IR2 = src.y */
|
||||
gte_mv_to_data_r(r_sz, C2_IR3), /* IR3 = src.z */
|
||||
nop2, /* COP2 transfer latency (2 slots) */
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(r_sx, C2_MAC1), /* MAC1 → r_sx (overwrites src.x with raw reciprocal-scaled) */
|
||||
gte_mv_from_data_r(r_sy, C2_MAC2),
|
||||
gte_mv_from_data_r(r_sz, C2_MAC3),
|
||||
shift_aright_var(r_sx, r_sx, r_shift),
|
||||
shift_aright_var(r_sy, r_sy, r_shift),
|
||||
shift_aright_var(r_sz, r_sz, r_shift),
|
||||
|
||||
/* ── I/O wrapper tail (~3 words) ───────────────────────────────────────────── */
|
||||
store_word(r_sx, r_dst, O_(V3_S4,x)),
|
||||
store_word(r_sy, r_dst, O_(V3_S4,y)),
|
||||
store_word(r_sz, r_dst, O_(V3_S4,z)),
|
||||
|
||||
/* ── atom_reads(R_TapePtr) atom_writes(R_TapePtr) ────────────────────────── */
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
#pragma region Bsked Atoms
|
||||
|
||||
typedef Struct_(Binds_SetGteWorld) {
|
||||
M3_S2* transform;
|
||||
typedef Struct_(Binds_SetGteMT3S2S4) {
|
||||
MT3_S2S4* transform;
|
||||
};
|
||||
internal MipsAtom_(set_gte_world) atom_info(
|
||||
atom_bind(Binds_SetGteWorld)
|
||||
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
|
||||
atom_bind(Binds_SetGteMT3S2S4)
|
||||
, atom_reads(R_TapePtr)
|
||||
){
|
||||
/* Pop matrix address from tape into R_T3 ($11) */
|
||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
|
||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||
|
||||
+60
-19
@@ -161,6 +161,8 @@ enum {
|
||||
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
||||
gte_cmd_op = 0x0C, /* Outer Product */
|
||||
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
||||
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
|
||||
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
|
||||
|
||||
/* --- GTE Command Bit-Field Layout ---
|
||||
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
||||
@@ -171,17 +173,22 @@ enum {
|
||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
||||
*
|
||||
* Shifts/masks below are the *bit positions* and *bit widths* of each
|
||||
* configurable field, used by the ENC_GTE_CMD encoder.
|
||||
* Shifts/masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
|
||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
||||
*/
|
||||
|
||||
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
|
||||
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
|
||||
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
|
||||
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
||||
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
||||
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
|
||||
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
|
||||
|
||||
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
|
||||
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
|
||||
gte_shift_fake_cmd = 20,
|
||||
gte_width_fake_cmd = 5,
|
||||
gte_mask_fake_cmd = 0x1F,
|
||||
};
|
||||
|
||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||
@@ -243,10 +250,10 @@ enum { _C2_OPS_ = 0
|
||||
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||
*
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||
* (which target the data register file on any coprocessor).
|
||||
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
||||
* and so the encoding lives next to its only consumer (this header).
|
||||
* and so the encoding is next to its only consumer (this header).
|
||||
*
|
||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
||||
enum { _C2_TX_SUBS_ = 0
|
||||
@@ -309,23 +316,24 @@ enum { _C2_TX_SUBS_ = 0
|
||||
|
||||
/* GTE Command Format
|
||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||
* The lower 25 bits are the GTE-specific command payload.
|
||||
* Lower 25 bits are GTE-specific command payload.
|
||||
*
|
||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||
* The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
||||
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
||||
*
|
||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go.
|
||||
* `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
|
||||
* It just ORs the per-field encoders together. */
|
||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||
|
||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
||||
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
||||
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
||||
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
||||
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
||||
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
|
||||
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
||||
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
||||
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
||||
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
||||
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
||||
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd ) << gte_shift_cmd )
|
||||
#define enc_gte_fake_cmd(x) (((x) & gte_mask_fake_cmd) << gte_shift_fake_cmd)
|
||||
|
||||
/* Composite: all six GTE fields + the COP2/CO base. */
|
||||
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
||||
@@ -363,11 +371,11 @@ enum { _C2_TX_SUBS_ = 0
|
||||
* (the perspective divide happens regardless of `sf`).
|
||||
*
|
||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
||||
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops —
|
||||
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
|
||||
* The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||
* `nclip` ends up wrong, and the triangle is culled.
|
||||
*
|
||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
|
||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
|
||||
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||
* --------------------------------------------------------------------------
|
||||
*/
|
||||
@@ -378,11 +386,45 @@ enum { _C2_TX_SUBS_ = 0
|
||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
|
||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */
|
||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
|
||||
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
||||
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||
|
||||
/* SQR / GPF cosmetic-bits compat helpers.
|
||||
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
||||
* The hardware ignores these bits (per PSX-SPX line 48). */
|
||||
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
|
||||
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
|
||||
|
||||
/* SQR — Square Vector.
|
||||
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
|
||||
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
|
||||
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
|
||||
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
|
||||
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
|
||||
* bit 19 sf=0
|
||||
* bit 10 lm=1
|
||||
* bits 5-0 cmd=0x28=SQR
|
||||
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
|
||||
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
|
||||
|
||||
/* GPF — General-purpose Interpolation.
|
||||
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
|
||||
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf*12)
|
||||
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
|
||||
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
|
||||
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
|
||||
* bit 19 sf=0
|
||||
* bit 10 lm=0
|
||||
* bits 5-0 cmd=0x3D=GPF
|
||||
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
||||
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
||||
|
||||
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
||||
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
||||
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
|
||||
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
|
||||
|
||||
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
||||
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
||||
@@ -433,7 +475,6 @@ enum {
|
||||
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
||||
|
||||
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
||||
*
|
||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
||||
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
||||
*
|
||||
|
||||
+85
-77
@@ -12,68 +12,57 @@
|
||||
#endif
|
||||
|
||||
#pragma region Tape Drive
|
||||
/* -----------------------------------------------------------------------------
|
||||
/* -----------------------------------------------------------------------------------------------------------
|
||||
* TAPE DRIVE ABI
|
||||
* -----------------------------------------------------------------------------
|
||||
* Note(Ed): One of the main purposes of this codebase is to help me
|
||||
* learn this, as such the information below may be entirely realized
|
||||
* or finalized conceptually.
|
||||
* -----------------------------------------------------------------------------
|
||||
* This ABI and its associated legos were directly inspired by researching
|
||||
* the work of Timothy Lottes and Onat Türkçüoğlu; along with many others.
|
||||
* It's the simplest bootstrap of a a directly executed chain of assemby
|
||||
* arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||
* These eventually lead to a terminal atom for the tape which is defined
|
||||
* below as "tape_exit".
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
|
||||
* as such the information below may not* be entirely realized or finalized conceptually.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* This ABI and its associated legos were directly inspired by researching the work of
|
||||
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
|
||||
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
||||
*
|
||||
* This behaves as one of the simplest runtime harnesses ontop of a
|
||||
* host-enviornment's execution engine to author and compose programs with.
|
||||
* From here various conventions can be further applied.
|
||||
* To make things easier to understand it may be better to focus on what this
|
||||
* ABI does not have. It does not have have any branching within the tape but
|
||||
* relative branches between atoms. Branching nearly is always downstream.
|
||||
* Stack usage is non-existent. Push/Pop, FIFO, or Arena/Bump data structures
|
||||
* are used by atoms explicitly. In it's current form withe C11 macro dsl,
|
||||
* the user also has to do manual register allocation per atom.
|
||||
* This behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
||||
* to author and compose programs with. From here various conventions can be further applied.
|
||||
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
||||
* Branching nearly is always downstream. Stack usage is non-existent.
|
||||
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
||||
* In it's current form with the C11 macro dsl, the user also has fullfill manual register allocation per atom.
|
||||
*
|
||||
* One of the remarkable things about utilizing this abi is its essentially
|
||||
* interopable with CPUs, GPUs, FPGA, or, basically anything
|
||||
* from the 5th generation consoles and onward.
|
||||
* The ABI directly reflects how all computational hardware must be architected
|
||||
* in order to execute digital logic effectively on current era tech.
|
||||
* On the PS1 we don't have access to a few features like multi-threading,
|
||||
* speculative execution, or L3 cache; but, we can set the foundation for legoing
|
||||
* whats required baseline wise for eventually expanding the harness and core atoms
|
||||
* to take those newer hardware features into account. For example, you can easily
|
||||
* expand this to support wave-based execution model on a PS2 or PS3.
|
||||
* Not having a stack or automatic register allocation means the user can't ignore
|
||||
* excessive argument shuffle across workload or waves and thier phases.
|
||||
* Crossing ABI boundaries to other runtimes that do has an obviouss penalties.
|
||||
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
|
||||
* or, basically anything from the 5th generation consoles and onward.
|
||||
* The ABI directly reflects how all computational hardware must be architected in order to execute
|
||||
* digital logic effectively on current era tech.
|
||||
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
|
||||
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
||||
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
||||
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
||||
* automatic register allocation means the user cannott ignore excessive argument shuffle across workload or
|
||||
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
||||
*
|
||||
* Learning data-oreinted code becomes a natural progression. Your not fighting
|
||||
* a stack-based procedural paradigm that wants to argument shuffle on the stack
|
||||
* by lack of constraints on how the user may "call" a procedure. The user doesn't
|
||||
* have to hammer down "rules" or patterns to know how to massage the compiler
|
||||
* to get the asesmbly into its natural form. The form is obvious, and once
|
||||
* the user gets to author their compoonents it becomes a game of tetris.
|
||||
* Learning data-oreinted code becomes a natural progression. Your not fighting a stack-based procedural
|
||||
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
||||
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
||||
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
||||
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
|
||||
* it becomes a game of tetris.
|
||||
*
|
||||
* Another feature is this ABI is very compatible with bootstrapping and developing
|
||||
* simple toolchains built off of bit-packed annotated command streams the user can
|
||||
* directly author, maintatain, and immediately execute. That being a color forth.
|
||||
* This can make the tetris less of a chore with some helpful policy generation for
|
||||
* allocation of registers, helping to choose resuable components, designing DSL on
|
||||
* the fly, etc.
|
||||
* -----------------------------------------------------------------------------
|
||||
* TODO(Ed): We ned pretty ascii diagrams and proper guides, articles, etc.
|
||||
* -----------------------------------------------------------------------------
|
||||
* For now this thing is just functioning and I'm abusing C11 + a lua metaprogram
|
||||
* to help establish a hybrid toolchain to ideate on a traditional text-based
|
||||
* authoring UX for this paradigm.
|
||||
* If pcsx-redux gets me viable hot-reload and persistent data storage beyond
|
||||
* save-states (just copying ram to filesystem). I can author a color forth to
|
||||
* mess around with, with an editor in-emulator or on the actual machine itself.
|
||||
* Assembly is tedius, but I think this codebase most likely has some of the most,
|
||||
* ergonomic you can come across..
|
||||
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
|
||||
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
|
||||
* That being like a color forth, or maybe something more familar like an immediate mode library
|
||||
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
|
||||
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
|
||||
* -----------------------------------------------------------------------------------------------------------
|
||||
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
|
||||
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
|
||||
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
|
||||
* (just copying ram to filesystem), I can author a color forth to mess around with.
|
||||
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
|
||||
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
|
||||
* */
|
||||
/* Register Allocation Info */
|
||||
enum {
|
||||
@@ -117,25 +106,32 @@ typedef Slice_(MipsCode);
|
||||
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
|
||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for atoms with value-args
|
||||
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||
// expands to:
|
||||
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; }
|
||||
#define MipsAtom_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); }
|
||||
|
||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||
// MipsAtomComp_(ac_X) { body }
|
||||
// expands to:
|
||||
// MipsCode ac_X[] align_(4) = { body };
|
||||
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||
|
||||
// Used for components with value-args (e.g., ac_format_f3_color).
|
||||
// FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
|
||||
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ac_X, ab, { body })
|
||||
// expands to:
|
||||
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
||||
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
|
||||
|
||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the
|
||||
file contains line-numbered content. Files containing only:
|
||||
- `MipsAtomComp_` static-array declarations, or
|
||||
- `MipsAtomComp_Proc_` (force-inline) function bodies whose line info gets
|
||||
attributed to the call site at the include point are otherwise omitted from the file table,
|
||||
which breaks the DWARF injection when it tries to resolve atom-component provenance paths.
|
||||
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
|
||||
// MipsCode ac_X[] align_(4) = { body };
|
||||
// atombuilder_unroll(ab, slice_from_array(MipsCode, ac_X));
|
||||
// }
|
||||
// The body must NOT include mac_yield() (the parent atom yields).
|
||||
// Inline-only callers (the generated `mac_<name>` aliases) skip this arg via metaprogram filtering;
|
||||
// escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
||||
#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(ab, slice_from_array(MipsCode, sym)); }
|
||||
|
||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||
Files containing only atoms and atom components.
|
||||
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||
The macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
||||
The constant is in `.rodata` and unreferenced; the linker may eliminate it.
|
||||
@@ -192,11 +188,13 @@ FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start
|
||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
||||
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
||||
|
||||
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), tb->used); tb->used += atoms.len; }
|
||||
|
||||
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||
@@ -231,7 +229,6 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
||||
add_ui_self(R_TapePtr, S_(MipsCode)),
|
||||
jump_reg( R_AtomJmp), nop,
|
||||
};
|
||||
|
||||
#pragma endregion Macro Atom Components
|
||||
|
||||
#pragma region Mips Atom Builder
|
||||
@@ -244,22 +241,33 @@ typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 u
|
||||
// Whatever the builder is writting to should most likely coresspond
|
||||
// to something that can fit within instruction cache?
|
||||
|
||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
|
||||
assert(ab->capacity - ab->used - code->len);
|
||||
mem_copy(ab->start, u4_(code->ptr), code->len);
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
|
||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode code) {
|
||||
assert(ab->capacity - ab->used - code.len);
|
||||
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot (arena accumulation) */
|
||||
mem_copy(u4_(dest), u4_(code.ptr), code.len);
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, code.len);
|
||||
}
|
||||
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||
|
||||
// When done authoring, utilize this to cap-off the atom
|
||||
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
||||
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
||||
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield));
|
||||
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot */
|
||||
mem_copy(u4_(dest), u4_(ac_yield), S_(ac_yield));
|
||||
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
|
||||
}
|
||||
|
||||
#define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used}
|
||||
#define mipsatom_from_builder(ab) C_(MipsAtom*, (ab).start)
|
||||
|
||||
// tb_emit_builder(tb, ab) — emit the builder's atom into the tape and advance tb->used.
|
||||
// Thin wrapper around tb_emit(tb, mipsatom_from_builder(ab[0])).
|
||||
// Equivalent to tb_emit(tb, code_<name>) for runtime-built atoms.
|
||||
FI_ void tb_emit_builder(TapeBuilder_R tb, MipsAtomBuilder_R ab) { tb_emit(tb, mipsatom_from_builder(ab[0])); }
|
||||
#pragma endregion Mips Atom Builder
|
||||
|
||||
#pragma region Mips Atom Procs
|
||||
|
||||
#pragma endregion Mips Atom Procs
|
||||
|
||||
#pragma region Baked Mips Atoms
|
||||
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
||||
|
||||
|
||||
+21
-3
@@ -9,17 +9,35 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Component)
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v2s2(U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, {
|
||||
FI_ Slice_MipsCode ac_load_v2s2(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, {
|
||||
load_half( rs_x, r_base, O_(V3_S2,x)),
|
||||
load_half( rs_y, r_base, O_(V3_S2,y)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, {
|
||||
FI_ Slice_MipsCode ac_store_v2s2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, {
|
||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
|
||||
FI_ Slice_MipsCode ac_load_v3s4(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, {
|
||||
load_word( rs_x, r_base, O_(V3_S4,x)),
|
||||
load_word( rs_y, r_base, O_(V3_S4,y)),
|
||||
load_word( rs_z, r_base, O_(V3_S4,z)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v3s4(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, {
|
||||
store_word(rt_x, base, offset + O_(V3_S4,x)),
|
||||
store_word(rt_y, base, offset + O_(V3_S4,y)),
|
||||
store_word(rt_z, base, offset + O_(V3_S4,z)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_v3s4(MipsAtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, {
|
||||
sub_s(rds_x, rds_x, rt_x),
|
||||
sub_s(rds_y, rds_y, rt_y),
|
||||
sub_s(rds_z, rds_z, rt_z),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rects2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, {
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||
|
||||
+55
-5
@@ -7,6 +7,18 @@
|
||||
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
||||
#define clamp_bot(X, B) max(X, B)
|
||||
|
||||
/* Convention
|
||||
<Type> ## <Width> _ <Component Type> ## <Component Width>
|
||||
For types with compound data (Ex: Rotation Matrix & Translation):
|
||||
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
|
||||
|
||||
A: Array
|
||||
V: Vector
|
||||
R: Range
|
||||
M: Matrix
|
||||
T: Translation
|
||||
*/
|
||||
|
||||
enum {
|
||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||
};
|
||||
@@ -26,23 +38,38 @@ typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
||||
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
|
||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
|
||||
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||
|
||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
||||
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
|
||||
typedef V3_S4 P3_S4;
|
||||
|
||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
|
||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
|
||||
|
||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||
|
||||
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
||||
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
|
||||
|
||||
/* RGA(Lengyel) reserved names (deferred):
|
||||
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
|
||||
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
|
||||
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
|
||||
|
||||
typedef Array_(V2_U1, 2);
|
||||
typedef Array_(V2_S2, 2);
|
||||
typedef Array_(V2_S2, 3);
|
||||
typedef Array_(V2_S2, 4);
|
||||
|
||||
enum {
|
||||
fp_one = (1 << 12),
|
||||
};
|
||||
|
||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||
|
||||
#define v2s2(x,y) (V2_S2){x,y}
|
||||
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
||||
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
||||
@@ -61,5 +88,28 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[2] += b[2] >> 1;
|
||||
}
|
||||
|
||||
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[0] -= b[0];
|
||||
(out_a[0])[1] -= b[1];
|
||||
(out_a[0])[2] -= b[2];
|
||||
}
|
||||
|
||||
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[0] -= b[0] >> 1;
|
||||
(out_a[0])[1] -= b[1] >> 1;
|
||||
(out_a[0])[2] -= b[2] >> 1;
|
||||
}
|
||||
|
||||
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||
(out_a[0])[0] *= b[0];
|
||||
(out_a[0])[1] *= b[1];
|
||||
(out_a[0])[2] *= b[2];
|
||||
}
|
||||
|
||||
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||
|
||||
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||
|
||||
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "gen/macs.h"
|
||||
# include "gen/offsets.h"
|
||||
# include "bios.h"
|
||||
# include "lottes_tape.h"
|
||||
#endif
|
||||
|
||||
@@ -8,11 +9,6 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
enum {
|
||||
bios_flushcache = 0x44,
|
||||
bios_table_addr = 0xA0,
|
||||
};
|
||||
|
||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||
|
||||
+12
-11
@@ -348,6 +348,12 @@ enum { _BitOffsets = 0
|
||||
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
||||
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
||||
|
||||
/* Shift Variable — register-shift forms.
|
||||
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
|
||||
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
|
||||
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
|
||||
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
|
||||
|
||||
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
||||
|
||||
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
||||
@@ -366,20 +372,18 @@ enum { _BitOffsets = 0
|
||||
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
||||
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
||||
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
||||
*
|
||||
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
||||
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
||||
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
||||
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
|
||||
*/
|
||||
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
||||
|
||||
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
||||
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`.
|
||||
*/
|
||||
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
|
||||
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
||||
|
||||
/* call_addr off — jump-and-link to immediate address.
|
||||
*
|
||||
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
||||
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
||||
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
||||
@@ -397,13 +401,7 @@ enum { _BitOffsets = 0
|
||||
* sub_s / sub_u → sub / subu
|
||||
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
||||
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
||||
*
|
||||
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
|
||||
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
|
||||
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
|
||||
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
|
||||
#undef add_s
|
||||
#undef sub_s
|
||||
*/
|
||||
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
||||
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
||||
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
||||
@@ -458,6 +456,9 @@ enum { _BitOffsets = 0
|
||||
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
||||
#define nop2 nop, nop
|
||||
|
||||
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
||||
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||
|
||||
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||
|
||||
|
||||
+84
-73
@@ -9,6 +9,34 @@
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom Components)
|
||||
|
||||
FI_ Slice_MipsCode ac_pad_set_centered_axes(MipsAtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, {
|
||||
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF),
|
||||
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF),
|
||||
store_word( r_scratch, r_state, O_(PadState,axes)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_pad_set_id_byte(MipsAtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, {
|
||||
add_ui( r_id, R_0, id_value),
|
||||
store_byte(r_id, r_state, O_(PadState,id)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, {
|
||||
add_ui( r_tmp, R_0, pad_status),
|
||||
store_word(r_tmp, r_state, O_(PadState,status)),
|
||||
})
|
||||
|
||||
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
||||
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
||||
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
||||
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(MipsAtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, {
|
||||
nor_u( r_buttons, r_buttons, R_0),
|
||||
store_half( r_buttons, r_pad_state, O_(PadState, buttons)),
|
||||
})
|
||||
|
||||
#pragma endregion MACs (Mips Atom Components)
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
/* ----- pad_bios_snapshot -----
|
||||
@@ -35,7 +63,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
||||
*/
|
||||
enum {
|
||||
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||
R_PadState = R_T1 atom_reg,
|
||||
R_PadState = R_T1 atom_reg atom_type(PadState*),
|
||||
R_RawStatus = R_T2 atom_reg,
|
||||
R_RawId = R_T3 atom_reg,
|
||||
};
|
||||
@@ -44,8 +72,8 @@ typedef Struct_(Binds_PadBiosSnapshot) {
|
||||
PadState* state;
|
||||
};
|
||||
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr)
|
||||
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||
) {
|
||||
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||
@@ -53,111 +81,97 @@ internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||
|
||||
/* === Read raw[0] (status) + raw[1] (id) */
|
||||
load_byte_u(R_RawStatus, R_PadRaw, 0),
|
||||
load_byte_u(R_RawId, R_PadRaw, 1),
|
||||
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
|
||||
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
|
||||
|
||||
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||
|
||||
atom_label(disconnected) /* === Disconnected body. */
|
||||
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
|
||||
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
|
||||
jump_rel(atom_offset(disconnected, snap_end)),
|
||||
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||
* The unconditional branch always jumps to snap_end, where mac_yield_tail()
|
||||
* transfers control to R_AtomJmp without re-loading it. */
|
||||
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
|
||||
mac_yield_load(),
|
||||
atom_label(skip_disconnected)
|
||||
|
||||
/* === Case 2: Pending (status == 0 && id == 0)
|
||||
* Combined check: if (status | id) != 0 then skip to id_dispatch.
|
||||
* Falls through to the Pending case only when both are zero. */
|
||||
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
|
||||
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui — harmless. */
|
||||
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
|
||||
|
||||
atom_label(pending) /* === Pending body */
|
||||
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
|
||||
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||
store_byte(R_RawId, R_PadState, O_(PadState,id)),
|
||||
jump_rel(atom_offset(pending, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||
|
||||
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */
|
||||
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)),
|
||||
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5
|
||||
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */
|
||||
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080),
|
||||
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
store_word( R_T5, R_PadState, O_(PadState,left_x)),
|
||||
add_ui( R_T4, R_0, 0x41),
|
||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
|
||||
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
|
||||
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
|
||||
load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
|
||||
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
|
||||
|
||||
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||
|
||||
atom_label(analog_stick) /* === AnalogStick body
|
||||
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10).
|
||||
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */
|
||||
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */
|
||||
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */
|
||||
store_byte( R_T5, R_PadState, O_(PadState,id)),
|
||||
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
|
||||
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
|
||||
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
|
||||
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
|
||||
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
|
||||
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||
mac_yield_load(),
|
||||
|
||||
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||
and_i( R_T4, R_RawId, 0xF0),
|
||||
add_ui( R_T5, R_0, 0x70),
|
||||
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
|
||||
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
|
||||
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||
|
||||
atom_label(analog_pad) /* === AnalogPad body
|
||||
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */
|
||||
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */
|
||||
store_word( R_T4, R_PadState, O_(PadState,status)),
|
||||
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */
|
||||
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */
|
||||
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */
|
||||
store_half( R_T4, R_PadState, O_(PadState,buttons)),
|
||||
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */
|
||||
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */
|
||||
store_half( R_T4, R_PadState, O_(PadState,right_x)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState,id)),
|
||||
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
|
||||
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
|
||||
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
|
||||
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
|
||||
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
|
||||
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||
store_byte( R_RawId, R_PadState, O_(PadState, id)),
|
||||
|
||||
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||
mac_yield_load(),
|
||||
@@ -166,11 +180,8 @@ atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the
|
||||
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
|
||||
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
|
||||
store_word( R_T4, R_PadState, O_(PadState,left_x)),
|
||||
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
|
||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
|
||||
/* Fall through to snap_end. */
|
||||
|
||||
atom_label(no_jump_fallthrough)
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# include "dsl.h"
|
||||
# include "gcc_asm.h"
|
||||
# include "mips.h"
|
||||
# include "bios.h"
|
||||
# include "pad.h"
|
||||
#endif
|
||||
|
||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
|
||||
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||
*
|
||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
|
||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||
*
|
||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
|
||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
{
|
||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||
(void)p0; (void)p1;
|
||||
|
||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||
// Use enums.
|
||||
|
||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||
* $a3 = 0x22 (immediate)
|
||||
* $t1 = 0x12 (function number)
|
||||
* $t2 = 0xB0 (BIOS B-table address) */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
||||
add_ui( rarg_1, rdiscard, bios_pad_buffer_size), /* $a1 = 0x22 */
|
||||
add_ui( rarg_3, rdiscard, bios_pad_buffer_size), /* $a3 = 0x22 */
|
||||
add_ui( rtmp_1, rdiscard, bios_init_pad_2), /* $t1 = 0x12 */
|
||||
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_rpins, r_use(p0), r_use(p1)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
|
||||
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||
u1_v(raw0)[0] = 0xFF;
|
||||
u1_v(raw1)[0] = 0xFF;
|
||||
|
||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
add_ui( rtmp_1, rdiscard, bios_start_pad_2), /* $t1 = 0x13 */
|
||||
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
}
|
||||
+79
-37
@@ -1,28 +1,30 @@
|
||||
#ifdef INTELLISENSE_DIRECTIVES
|
||||
# pragma once
|
||||
# include "dsl.h"
|
||||
# include "math.h"
|
||||
#endif
|
||||
|
||||
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||
* Wire is active-low (0 = pressed).
|
||||
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */
|
||||
enum {
|
||||
Bit_(Pad_Select, 0),
|
||||
Bit_(Pad_L3, 1),
|
||||
Bit_(Pad_R3, 2),
|
||||
Bit_(Pad_Start, 3),
|
||||
Bit_(Pad_Up, 4),
|
||||
Bit_(Pad_Right, 5),
|
||||
Bit_(Pad_Down, 6),
|
||||
Bit_(Pad_Left, 7),
|
||||
Bit_(Pad_L2, 8),
|
||||
Bit_(Pad_R2, 9),
|
||||
Bit_(Pad_L1, 10),
|
||||
Bit_(Pad_R1, 11),
|
||||
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
|
||||
* active-low-to-active-high inversion is applied bit-by-bit. */
|
||||
typedef Enum_(U2, PadBtns) {
|
||||
Bit_(Pad_Select, 0),
|
||||
Bit_(Pad_L3, 1),
|
||||
Bit_(Pad_R3, 2),
|
||||
Bit_(Pad_Start, 3),
|
||||
Bit_(Pad_Up, 4),
|
||||
Bit_(Pad_Right, 5),
|
||||
Bit_(Pad_Down, 6),
|
||||
Bit_(Pad_Left, 7),
|
||||
Bit_(Pad_L2, 8),
|
||||
Bit_(Pad_R2, 9),
|
||||
Bit_(Pad_L1, 10),
|
||||
Bit_(Pad_R1, 11),
|
||||
Bit_(Pad_Triangle, 12),
|
||||
Bit_(Pad_Circle, 13),
|
||||
Bit_(Pad_Cross, 14),
|
||||
Bit_(Pad_Square, 15),
|
||||
Bit_(Pad_Circle, 13),
|
||||
Bit_(Pad_Cross, 14),
|
||||
Bit_(Pad_Square, 15),
|
||||
};
|
||||
|
||||
enum {
|
||||
@@ -32,18 +34,22 @@ enum {
|
||||
Pad1 = 1 << PadId_Offset,
|
||||
};
|
||||
|
||||
#define pad0_(btn_id) (btn_id << Pad0)
|
||||
#define pad1_(btn_id) (btn_id << Pad1)
|
||||
|
||||
/* ============================================================
|
||||
/* =============================================================================
|
||||
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
||||
* ============================================================ */
|
||||
* ============================================================================= */
|
||||
|
||||
enum {
|
||||
PAD_BIOS_RAW_SIZE = 0x22,
|
||||
};
|
||||
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
|
||||
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
|
||||
typedef Struct_(PadBiosRaw) {
|
||||
U1 bytes[PAD_BIOS_RAW_SIZE];
|
||||
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
|
||||
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
|
||||
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
|
||||
V2_U1 right; /* offset 4-5 (right stick x, y) */
|
||||
V2_U1 left; /* offset 6-7 (left stick x, y) */
|
||||
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
|
||||
};
|
||||
|
||||
typedef Enum_(U4, PadStatus) {
|
||||
@@ -56,18 +62,54 @@ typedef Enum_(U4, PadStatus) {
|
||||
PadStatus_Invalid,
|
||||
};
|
||||
|
||||
/* PadState — per-port normalized runtime state.
|
||||
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||
* The struct size stays 12 bytes (unchanged from the prior order,
|
||||
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */
|
||||
typedef Struct_(PadState) {
|
||||
PadStatus status; /* offset 0, size 4 (U4) */
|
||||
U2 buttons; /* offset 4, size 2 */
|
||||
U1 id; /* offset 6, size 1 */
|
||||
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
|
||||
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
|
||||
U1 left_y; /* offset 9, size 1 */
|
||||
U1 right_x; /* offset 10, size 1 */
|
||||
U1 right_y; /* offset 11, size 1 */
|
||||
/* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
|
||||
* PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
|
||||
* when the controller id does not match any known controller type.
|
||||
* PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
|
||||
* four byte axes at PadState.left_x through PadState.right_y. */
|
||||
typedef Enum_(U1, PadRawStatus) {
|
||||
PadRawStatus_Ok = 0x00,
|
||||
PadRawStatus_Timeout = 0xFF,
|
||||
};
|
||||
typedef Enum_(U1, PadRawId) {
|
||||
PadRawId_Digital = 0x41,
|
||||
PadRawId_AnalogStick = 0x53,
|
||||
PadRawId_AnalogPadMask = 0xF0,
|
||||
PadRawId_AnalogPadValue = 0x70,
|
||||
};
|
||||
typedef Enum_(U1, PadUnknownId) {
|
||||
PadUnknownId_Sentinel = 0xFF,
|
||||
};
|
||||
typedef Enum_(U4, PadAxisCentered) {
|
||||
PadAxis_Centered_Hi = 0x8080,
|
||||
PadAxis_Centered_Lo = 0x8080,
|
||||
PadAxis_Centered_Word = 0x80808080U,
|
||||
};
|
||||
typedef Enum_(U1, PadDeadZone) {
|
||||
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
|
||||
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
|
||||
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
|
||||
};
|
||||
|
||||
|
||||
typedef Struct_(PadAxes) {
|
||||
V2_U1 left; /* offset 8-9 */
|
||||
V2_U1 right; /* offset 10-11 */
|
||||
};
|
||||
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||
typedef Struct_(PadState) {
|
||||
PadStatus status; /* offset 0, (U4) */
|
||||
PadBtns buttons; /* offset 4, */
|
||||
U1 id; /* offset 6, */
|
||||
byte_pad(1); /* offset 7, explicit pad to align the axes block */
|
||||
union {
|
||||
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
|
||||
struct {
|
||||
V2_U1 left; /* offset 8-9 */
|
||||
V2_U1 right; /* offset 10-11 */
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
|
||||
|
||||
+23
-5
@@ -64,9 +64,9 @@ typedef Struct_(Tile) {
|
||||
Linear Algebra
|
||||
*/
|
||||
|
||||
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix");
|
||||
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix");
|
||||
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
|
||||
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
|
||||
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||
|
||||
// Rotation, Translation, Perspective
|
||||
|
||||
@@ -99,5 +99,23 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
||||
);
|
||||
}
|
||||
|
||||
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix");
|
||||
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix");
|
||||
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
|
||||
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
|
||||
|
||||
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
|
||||
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
|
||||
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
||||
|
||||
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
|
||||
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
|
||||
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
|
||||
|
||||
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
|
||||
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
|
||||
|
||||
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
|
||||
|
||||
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
|
||||
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
|
||||
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
|
||||
|
||||
|
||||
@@ -54,6 +54,15 @@ WORD_COUNT(gte_sw, 1)
|
||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||
WORD_COUNT(gte_cmdw_sqr, 1)
|
||||
WORD_COUNT(gte_cmdw_gpf, 1)
|
||||
WORD_COUNT(shift_lleft_var, 1)
|
||||
WORD_COUNT(shift_aright_var, 1)
|
||||
WORD_COUNT(li_s, 1)
|
||||
WORD_COUNT(and_i, 1)
|
||||
WORD_COUNT(add_si, 1)
|
||||
WORD_COUNT(branch_lt_zero, 1)
|
||||
WORD_COUNT(sub_s, 1)
|
||||
WORD_COUNT(sub_u, 1)
|
||||
WORD_COUNT(nop2, 2)
|
||||
|
||||
|
||||
@@ -39,3 +39,350 @@ WORD_COUNT(mac_put_disp_env, 5)
|
||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||
WORD_COUNT(mac_put_draw_env, 16)
|
||||
|
||||
#define mac_resolve_look_at__input_and_sub(...) \
|
||||
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)) \
|
||||
, load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)) \
|
||||
, load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)) \
|
||||
, add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)) \
|
||||
, load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtScratch,scratch_base)) \
|
||||
, add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtScratch)) /* Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation
|
||||
* column). Reuse r_tmp0/r_tmp1/r_tmp2. Offsets via O_(ResolveLookAtScratch,*). */ \
|
||||
, load_word(r_tmp0, r_eye_ptr, O_(P3_S4,x)) \
|
||||
, load_word(r_tmp1, r_eye_ptr, O_(P3_S4,y)) \
|
||||
, load_word(r_tmp2, r_eye_ptr, O_(P3_S4,z)) \
|
||||
, nop /* load-delay */ \
|
||||
, store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,eye.x)) \
|
||||
, store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,eye.y)) \
|
||||
, store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye.z)) /* Stage up_in.x/y/z into the scratchpad (atom 2 reads these for the outer
|
||||
* product with uz). Reuse r_tmp0/r_tmp1/r_tmp2. */ \
|
||||
, load_word(r_tmp0, r_up_in_ptr, O_(V3_S4,x)) \
|
||||
, load_word(r_tmp1, r_up_in_ptr, O_(V3_S4,y)) \
|
||||
, load_word(r_tmp2, r_up_in_ptr, O_(V3_S4,z)) \
|
||||
, nop /* load-delay */ \
|
||||
, store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,up_in.x)) \
|
||||
, store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,up_in.y)) \
|
||||
, store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in.z)) /* Compute fwd = target - eye. */ \
|
||||
, load_word(r_tmp0, r_target_ptr, O_(P3_S4,x)) \
|
||||
, load_word(r_tmp1, r_target_ptr, O_(P3_S4,y)) \
|
||||
, load_word(r_tmp2, r_target_ptr, O_(P3_S4,z)) \
|
||||
, load_word(r_tmp3, r_eye_ptr, O_(P3_S4,x)) \
|
||||
, load_word(R_AT, r_eye_ptr, O_(P3_S4,y)) \
|
||||
, load_word(R_V0, r_eye_ptr, O_(P3_S4,z)) \
|
||||
, nop /* load-delay */ \
|
||||
, sub_u(r_tmp0, r_tmp0, r_tmp3) \
|
||||
, sub_u(r_tmp1, r_tmp1, R_AT) \
|
||||
, sub_u(r_tmp2, r_tmp2, R_V0) /* Store fwd.x/y/z (atom 1 reads these as the normalize src). */ \
|
||||
, store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,fwd.x)) \
|
||||
, store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,fwd.y)) \
|
||||
, store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd.z)) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__input_and_sub, 34)
|
||||
|
||||
#define mac_resolve_look_at__cross_uz_up_in_to_right(...) \
|
||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_g = &uz */ \
|
||||
, add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)) /* r_h = &up_in */ \
|
||||
, add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)) /* r_f = &right (out) */ \
|
||||
, nop /* Load a (uz).x/y/z into r_a/r_b/r_c. */ \
|
||||
, load_word(r_a, r_g, O_(V3_S4,x)) \
|
||||
, load_word(r_b, r_g, O_(V3_S4,y)) \
|
||||
, load_word(r_c, r_g, O_(V3_S4,z)) \
|
||||
, nop /* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (hardcoded; reusing the
|
||||
* body's last two loads is fine because the load-delay slot is the nop
|
||||
* after the third load, and mtc2 below doesn't read these regs). */ \
|
||||
, load_word(r_d, r_h, O_(V3_S4,x)) \
|
||||
, load_word(R_AT, r_h, O_(V3_S4,y)) \
|
||||
, load_word(R_V0, r_h, O_(V3_S4,z)) \
|
||||
, nop /* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */ \
|
||||
, gte_mv_to_data_r(r_a, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_b, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_c, C2_IR3) \
|
||||
, gte_mv_to_data_r(r_d, C2_VXY0) /* D1 = b.x */ \
|
||||
, gte_mv_to_data_r(R_AT, C2_VZ0) /* D2 = b.y */ \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY1) /* D3 = b.z */ \
|
||||
, nop2 /* MTC2 retirement (CPU→COP2 2-slot delay) */ \
|
||||
, gte_cmdw_outer_product /* OP fires; MAC1/2/3 = a × b */ /* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */ \
|
||||
, gte_mv_from_data_r(r_a, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_b, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_c, C2_MAC3) \
|
||||
, nop /* MFC2 retirement */ /* Store out.x/y/z to r_f (out ptr = scratch+32). */ \
|
||||
, store_word(r_a, r_f, O_(V3_S4,x)) \
|
||||
, store_word(r_b, r_f, O_(V3_S4,y)) \
|
||||
, store_word(r_c, r_f, O_(V3_S4,z)) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__cross_uz_up_in_to_right, 29)
|
||||
|
||||
#define mac_resolve_look_at__cross_uz_ux_to_up(...) \
|
||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_g = &uz */ \
|
||||
, add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)) /* r_h = &ux */ \
|
||||
, add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)) /* r_f = &up (out) */ \
|
||||
, nop /* Load a (uz).x/y/z into r_a/r_b/r_c. */ \
|
||||
, load_word(r_a, r_g, O_(V3_S4,x)) \
|
||||
, load_word(r_b, r_g, O_(V3_S4,y)) \
|
||||
, load_word(r_c, r_g, O_(V3_S4,z)) \
|
||||
, nop /* Load b (ux).x/y/z into r_d + R_AT/R_V0. */ \
|
||||
, load_word(r_d, r_h, O_(V3_S4,x)) \
|
||||
, load_word(R_AT, r_h, O_(V3_S4,y)) \
|
||||
, load_word(R_V0, r_h, O_(V3_S4,z)) \
|
||||
, nop /* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */ \
|
||||
, gte_mv_to_data_r(r_a, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_b, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_c, C2_IR3) \
|
||||
, gte_mv_to_data_r(r_d, C2_VXY0) \
|
||||
, gte_mv_to_data_r(R_AT, C2_VZ0) \
|
||||
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||
, nop2 \
|
||||
, gte_cmdw_outer_product \
|
||||
, gte_mv_from_data_r(r_a, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_b, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_c, C2_MAC3) \
|
||||
, nop \
|
||||
, store_word(r_a, r_f, O_(V3_S4,x)) \
|
||||
, store_word(r_b, r_f, O_(V3_S4,y)) \
|
||||
, store_word(r_c, r_f, O_(V3_S4,z)) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__cross_uz_ux_to_up, 29)
|
||||
|
||||
#define mac_resolve_look_at__normalize_fwd_to_uz(...) \
|
||||
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,fwd)) /* r_a = &fwd */ \
|
||||
, add_si(r_b, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_b = &uz */ \
|
||||
, nop /* Load src.x/y/z from r_a into r_e/r_f/r_i. */ \
|
||||
, load_word(r_e, r_a, O_(V3_S4,x)) \
|
||||
, load_word(r_f, r_a, O_(V3_S4,y)) \
|
||||
, load_word(r_i, r_a, O_(V3_S4,z)) \
|
||||
, nop /* load-delay */ /* Stage 1: mtc2 src → IR1/2/3, SQR fires. */ \
|
||||
, gte_mv_to_data_r(r_e, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_f, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_i, C2_IR3) \
|
||||
, nop \
|
||||
, gte_cmdw_sqr /* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */ \
|
||||
, gte_mv_from_data_r(r_d, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_g, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_recip_est, C2_MAC3) \
|
||||
, nop \
|
||||
, add_u(r_recip_est, r_recip_est, r_g) \
|
||||
, add_u(r_recip_est, r_recip_est, r_d) \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_LZCS) \
|
||||
, nop2 \
|
||||
, gte_mv_from_data_r(r_h, C2_LZCR) \
|
||||
, nop /* Stage 3: compute shift amount, align |v|² to bit 24. */ \
|
||||
, and_i( r_h, r_h, -2) \
|
||||
, li_s( r_shift, 31) \
|
||||
, sub_s( r_shift, r_shift, r_h) \
|
||||
, shift_aright(r_shift, r_shift, 1) \
|
||||
, add_si( r_a, r_h, -24) /* r_a = LZCR - 24 (overlapping with r_recip_est; src ptr no longer needed) */ \
|
||||
, branch_lt_zero(r_a, atom_offset(srav_path_fwd_to_uz, aligned_done_fwd_to_uz)) \
|
||||
, nop \
|
||||
, jump_rel(atom_offset(aligned_done_fwd_to_uz, srav_path_fwd_to_uz)) \
|
||||
, shift_lleft_var(r_recip_est, r_recip_est, r_a) \
|
||||
, atom_label(srav_path_fwd_to_uz) \
|
||||
, li_s( r_a, 24) \
|
||||
, sub_s( r_a, r_a, r_h) \
|
||||
, shift_aright_var(r_recip_est, r_recip_est, r_a) \
|
||||
, atom_label(aligned_done_fwd_to_uz) /* r_recip_est holds |v|² aligned to bit 24. */ \
|
||||
, add_si( r_recip_est, r_recip_est, -64) \
|
||||
, shift_lleft(r_recip_est, r_recip_est, 1) \
|
||||
, load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)) \
|
||||
, or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)) \
|
||||
, add_u(r_a, r_a, r_recip_est) \
|
||||
, load_half(r_recip_est, r_a, 0) \
|
||||
, nop /* Stage 4: GPF + srav finalize. */ \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||
, gte_mv_to_data_r(r_e, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_f, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_i, C2_IR3) \
|
||||
, nop2 \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(r_e, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_f, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_i, C2_MAC3) \
|
||||
, shift_aright_var(r_e, r_e, r_shift) \
|
||||
, shift_aright_var(r_f, r_f, r_shift) \
|
||||
, shift_aright_var(r_i, r_i, r_shift) /* Store result.x/y/z to r_b (dst ptr = scratch+16). */ \
|
||||
, store_word(r_e, r_b, O_(V3_S4,x)) \
|
||||
, store_word(r_f, r_b, O_(V3_S4,y)) \
|
||||
, store_word(r_i, r_b, O_(V3_S4,z)) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__normalize_fwd_to_uz, 59)
|
||||
|
||||
#define mac_resolve_look_at__normalize_right_to_ux(...) \
|
||||
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,right)) /* r_a = &right */ \
|
||||
, add_si(r_b, r_scratch, O_(ResolveLookAtScratch,ux)) /* r_b = &ux */ \
|
||||
, nop \
|
||||
, load_word(r_e, r_a, O_(V3_S4,x)) \
|
||||
, load_word(r_f, r_a, O_(V3_S4,y)) \
|
||||
, load_word(r_i, r_a, O_(V3_S4,z)) \
|
||||
, nop \
|
||||
, gte_mv_to_data_r(r_e, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_f, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_i, C2_IR3) \
|
||||
, nop \
|
||||
, gte_cmdw_sqr \
|
||||
, gte_mv_from_data_r(r_d, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_g, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_recip_est, C2_MAC3) \
|
||||
, nop \
|
||||
, add_u(r_recip_est, r_recip_est, r_g) \
|
||||
, add_u(r_recip_est, r_recip_est, r_d) \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_LZCS) \
|
||||
, nop2 \
|
||||
, gte_mv_from_data_r(r_h, C2_LZCR) \
|
||||
, nop \
|
||||
, and_i( r_h, r_h, -2) \
|
||||
, li_s( r_shift, 31) \
|
||||
, sub_s( r_shift, r_shift, r_h) \
|
||||
, shift_aright(r_shift, r_shift, 1) \
|
||||
, add_si( r_a, r_h, -24) \
|
||||
, branch_lt_zero(r_a, atom_offset(srav_path_right_to_ux, aligned_done_right_to_ux)) \
|
||||
, nop \
|
||||
, jump_rel(atom_offset(aligned_done_right_to_ux, srav_path_right_to_ux)) \
|
||||
, shift_lleft_var(r_recip_est, r_recip_est, r_a) \
|
||||
, atom_label(srav_path_right_to_ux) \
|
||||
, li_s( r_a, 24) \
|
||||
, sub_s( r_a, r_a, r_h) \
|
||||
, shift_aright_var(r_recip_est, r_recip_est, r_a) \
|
||||
, atom_label(aligned_done_right_to_ux) \
|
||||
, add_si( r_recip_est, r_recip_est, -64) \
|
||||
, shift_lleft(r_recip_est, r_recip_est, 1) \
|
||||
, load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)) \
|
||||
, or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)) \
|
||||
, add_u(r_a, r_a, r_recip_est) \
|
||||
, load_half(r_recip_est, r_a, 0) \
|
||||
, nop \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||
, gte_mv_to_data_r(r_e, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_f, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_i, C2_IR3) \
|
||||
, nop2 \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(r_e, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_f, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_i, C2_MAC3) \
|
||||
, shift_aright_var(r_e, r_e, r_shift) \
|
||||
, shift_aright_var(r_f, r_f, r_shift) \
|
||||
, shift_aright_var(r_i, r_i, r_shift) \
|
||||
, store_word(r_e, r_b, O_(V3_S4,x)) \
|
||||
, store_word(r_f, r_b, O_(V3_S4,y)) \
|
||||
, store_word(r_i, r_b, O_(V3_S4,z)) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__normalize_right_to_ux, 59)
|
||||
|
||||
#define mac_resolve_look_at__normalize_up_to_uy(...) \
|
||||
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,up)) /* r_a = &up */ \
|
||||
, add_si(r_b, r_scratch, O_(ResolveLookAtScratch,uy)) /* r_b = &uy */ \
|
||||
, nop \
|
||||
, load_word(r_e, r_a, O_(V3_S4,x)) \
|
||||
, load_word(r_f, r_a, O_(V3_S4,y)) \
|
||||
, load_word(r_i, r_a, O_(V3_S4,z)) \
|
||||
, nop \
|
||||
, gte_mv_to_data_r(r_e, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_f, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_i, C2_IR3) \
|
||||
, nop \
|
||||
, gte_cmdw_sqr \
|
||||
, gte_mv_from_data_r(r_d, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_g, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_recip_est, C2_MAC3) \
|
||||
, nop \
|
||||
, add_u(r_recip_est, r_recip_est, r_g) \
|
||||
, add_u(r_recip_est, r_recip_est, r_d) \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_LZCS) \
|
||||
, nop2 \
|
||||
, gte_mv_from_data_r(r_h, C2_LZCR) \
|
||||
, nop \
|
||||
, and_i( r_h, r_h, -2) \
|
||||
, li_s( r_shift, 31) \
|
||||
, sub_s( r_shift, r_shift, r_h) \
|
||||
, shift_aright(r_shift, r_shift, 1) \
|
||||
, add_si( r_a, r_h, -24) \
|
||||
, branch_lt_zero(r_a, atom_offset(srav_path_up_to_uy, aligned_done_up_to_uy)) \
|
||||
, nop \
|
||||
, jump_rel(atom_offset(aligned_done_up_to_uy, srav_path_up_to_uy)) \
|
||||
, shift_lleft_var(r_recip_est, r_recip_est, r_a) \
|
||||
, atom_label(srav_path_up_to_uy) \
|
||||
, li_s( r_a, 24) \
|
||||
, sub_s( r_a, r_a, r_h) \
|
||||
, shift_aright_var(r_recip_est, r_recip_est, r_a) \
|
||||
, atom_label(aligned_done_up_to_uy) \
|
||||
, add_si( r_recip_est, r_recip_est, -64) \
|
||||
, shift_lleft(r_recip_est, r_recip_est, 1) \
|
||||
, load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)) \
|
||||
, or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)) \
|
||||
, add_u(r_a, r_a, r_recip_est) \
|
||||
, load_half(r_recip_est, r_a, 0) \
|
||||
, nop \
|
||||
, gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||
, gte_mv_to_data_r(r_e, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_f, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_i, C2_IR3) \
|
||||
, nop2 \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(r_e, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_f, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_i, C2_MAC3) \
|
||||
, shift_aright_var(r_e, r_e, r_shift) \
|
||||
, shift_aright_var(r_f, r_f, r_shift) \
|
||||
, shift_aright_var(r_i, r_i, r_shift) \
|
||||
, store_word(r_e, r_b, O_(V3_S4,x)) \
|
||||
, store_word(r_f, r_b, O_(V3_S4,y)) \
|
||||
, store_word(r_i, r_b, O_(V3_S4,z)) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__normalize_up_to_uy, 59)
|
||||
|
||||
#define mac_resolve_look_at__populate_and_translate(...) \
|
||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)) \
|
||||
, add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)) /* Compute the 4 scratch pointers in their dedicated GPRs. */ \
|
||||
, add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)) /* r_pux = &ux */ \
|
||||
, add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)) /* r_puy = &uy */ \
|
||||
, add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)) /* r_puz = &uz */ \
|
||||
, add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)) /* r_peye = &eye */ \
|
||||
, nop /* ── m[0] = (S2)ux ── */ \
|
||||
, load_word(r_tmp0, r_pux, O_(V3_S4,x)) \
|
||||
, load_word(r_tmp1, r_pux, O_(V3_S4,y)) \
|
||||
, load_word(r_tmp2, r_pux, O_(V3_S4,z)) \
|
||||
, nop \
|
||||
, store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])) \
|
||||
, store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])) \
|
||||
, store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])) /* ── m[1] = (S2)uy ── */ \
|
||||
, load_word(r_tmp0, r_puy, O_(V3_S4,x)) \
|
||||
, load_word(r_tmp1, r_puy, O_(V3_S4,y)) \
|
||||
, load_word(r_tmp2, r_puy, O_(V3_S4,z)) \
|
||||
, nop \
|
||||
, store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])) \
|
||||
, store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])) \
|
||||
, store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])) /* ── m[2] = (S2)uz ── */ \
|
||||
, load_word(r_tmp0, r_puz, O_(V3_S4,x)) \
|
||||
, load_word(r_tmp1, r_puz, O_(V3_S4,y)) \
|
||||
, load_word(r_tmp2, r_puz, O_(V3_S4,z)) \
|
||||
, nop \
|
||||
, store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])) \
|
||||
, store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])) \
|
||||
, store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])) /* ── Translation column t[i] = R * (-eye) ─────────────────────────────
|
||||
* pos = -eye: load eye.x/y/z from r_peye, negate via sub_u from R_0. */ \
|
||||
, load_word(r_tmp0, r_peye, O_(P3_S4,x)) \
|
||||
, load_word(r_tmp1, r_peye, O_(P3_S4,y)) \
|
||||
, load_word(r_tmp2, r_peye, O_(P3_S4,z)) \
|
||||
, nop \
|
||||
, sub_u(r_tmp0, R_0, r_tmp0) /* pos.x = -eye.x */ \
|
||||
, sub_u(r_tmp1, R_0, r_tmp1) \
|
||||
, sub_u(r_tmp2, R_0, r_tmp2) /* mtc2 IR1/2/3 = pos (for MVMVA — input vector registers). */ \
|
||||
, gte_mv_to_data_r(r_tmp0, C2_IR1) \
|
||||
, gte_mv_to_data_r(r_tmp1, C2_IR2) \
|
||||
, gte_mv_to_data_r(r_tmp2, C2_IR3) \
|
||||
, nop2 /* MVMVA: MAC1/2/3 = R * IR with cv=0 (no TR vector), mx=0 (rotation matrix),
|
||||
* sf=0 (no shift), v=0 (V0 = IR1/2/3, no far-plane clipping). The pre-set
|
||||
* rotation matrix is the one set by the preceding set_gte_world atom.
|
||||
* gte_cmdw_mvmva is parameterless and defaults to cv=0/mx=0/sf=0/v=0. */ \
|
||||
, gte_cmdw_mvmva \
|
||||
, nop /* GTE interlock */ /* mfc2 MAC1/2/3 → r_tmp0/r_tmp1/r_tmp2 (sign-extended into 32-bit GPRs).
|
||||
* MAC1/2/3 hold R*v with no TR add and no perspective divide — exactly the
|
||||
* 3 distinct world-space translation values we need for t[0..2]. */ \
|
||||
, gte_mv_from_data_r(r_tmp0, C2_MAC1) \
|
||||
, gte_mv_from_data_r(r_tmp1, C2_MAC2) \
|
||||
, gte_mv_from_data_r(r_tmp2, C2_MAC3) \
|
||||
, nop \
|
||||
, store_word(r_tmp0, r_look_at, O_(MT3_S2S4,t[0])) \
|
||||
, store_word(r_tmp1, r_look_at, O_(MT3_S2S4,t[1])) \
|
||||
, store_word(r_tmp2, r_look_at, O_(MT3_S2S4,t[2])) \
|
||||
, mac_yield()
|
||||
WORD_COUNT(mac_resolve_look_at__populate_and_translate, 50)
|
||||
|
||||
|
||||
@@ -8,7 +8,37 @@
|
||||
#pragma region hello_camera
|
||||
|
||||
|
||||
// --- atom: pad_apply_input (60 words) ---
|
||||
// --- atom: resolve_look_at__normalize_fwd_to_uz (62 words) ---
|
||||
|
||||
#define _atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz 6
|
||||
#define _atom_offset_aligned_done_fwd_to_uz_srav_path_fwd_to_uz 1
|
||||
|
||||
enum {
|
||||
atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz = _atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz,
|
||||
atom_offset_aligned_done_fwd_to_uz_srav_path_fwd_to_uz = _atom_offset_aligned_done_fwd_to_uz_srav_path_fwd_to_uz,
|
||||
};
|
||||
|
||||
// --- atom: resolve_look_at__normalize_right_to_ux (62 words) ---
|
||||
|
||||
#define _atom_offset_srav_path_right_to_ux_aligned_done_right_to_ux 6
|
||||
#define _atom_offset_aligned_done_right_to_ux_srav_path_right_to_ux 1
|
||||
|
||||
enum {
|
||||
atom_offset_srav_path_right_to_ux_aligned_done_right_to_ux = _atom_offset_srav_path_right_to_ux_aligned_done_right_to_ux,
|
||||
atom_offset_aligned_done_right_to_ux_srav_path_right_to_ux = _atom_offset_aligned_done_right_to_ux_srav_path_right_to_ux,
|
||||
};
|
||||
|
||||
// --- atom: resolve_look_at__normalize_up_to_uy (62 words) ---
|
||||
|
||||
#define _atom_offset_srav_path_up_to_uy_aligned_done_up_to_uy 6
|
||||
#define _atom_offset_aligned_done_up_to_uy_srav_path_up_to_uy 1
|
||||
|
||||
enum {
|
||||
atom_offset_srav_path_up_to_uy_aligned_done_up_to_uy = _atom_offset_srav_path_up_to_uy_aligned_done_up_to_uy,
|
||||
atom_offset_aligned_done_up_to_uy_srav_path_up_to_uy = _atom_offset_aligned_done_up_to_uy_srav_path_up_to_uy,
|
||||
};
|
||||
|
||||
// --- atom: pad_input_cube_rotation (60 words) ---
|
||||
|
||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||
@@ -26,6 +56,24 @@ enum {
|
||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||
};
|
||||
|
||||
// --- atom: pad_input_cam (40 words) ---
|
||||
|
||||
#define _atom_offset_left_x_exit_left_x 3
|
||||
#define _atom_offset_right_x_exit_right_x 3
|
||||
#define _atom_offset_up_y_exit_up_y 3
|
||||
#define _atom_offset_down_y_exit_down_y 3
|
||||
#define _atom_offset_cross_z_exit_cross_z 3
|
||||
#define _atom_offset_circle_z_exit_circle_z 3
|
||||
|
||||
enum {
|
||||
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
|
||||
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
|
||||
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
|
||||
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
|
||||
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
|
||||
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
||||
};
|
||||
|
||||
// --- atom: cube_g4_face (76 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||
|
||||
@@ -24,8 +24,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
|
||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
@@ -35,8 +35,8 @@ MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_draw_env, {
|
||||
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
|
||||
/*
|
||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||
* References:
|
||||
@@ -90,6 +90,676 @@ MipsAtomComp_Proc_(ac_put_draw_env, {
|
||||
|
||||
#pragma endregion MACs
|
||||
|
||||
#pragma region Atom Procs
|
||||
// Modular Atoms
|
||||
|
||||
enum {
|
||||
/* Wave-context GPR carrier for the resolve_look_at bundle: the scratch base.
|
||||
* Set by atom 0 (popped from tape), read by atoms 1-6 (used as pointer base).
|
||||
* Type is U4* — this holds the scratch base address (smem.scratchpad value).
|
||||
*
|
||||
* Other wave-context carriers (R_ResolveUzPtr / UxPtr / UyPtr) used in the
|
||||
* prior design were dropped: the new chain atoms compute their src/dst
|
||||
* addresses internally from R_ResolveScratch + hardcoded_offset. */
|
||||
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
|
||||
#define R_ResolveScratch_Code R_T4_Code
|
||||
};
|
||||
typedef Struct_(Binds_ResolveLookAt) {
|
||||
MT3_S2S4* look_at;
|
||||
P3_S4* eye;
|
||||
P3_S4* target;
|
||||
V3_S4* up_in;
|
||||
};
|
||||
|
||||
/* Per-atom bind-pop structs for the resolve_look_at bundle.
|
||||
* Atom 0 (input_and_sub) is the ONLY atom that touches the C-side pointers +
|
||||
* scratch base. Atoms 1-6 use scratch + hardcoded offsets internally.
|
||||
* Field types are U4 (raw pointer value) because the structs are populated
|
||||
* by the frame-time bundle helper with the literal C-side pointer values. */
|
||||
typedef Struct_(Binds_ResolveLookAtScratch) {
|
||||
U4 scratch_base; /* U4 (scratch base address — populated by helper with u4_(smem.scratchpad)) */
|
||||
};
|
||||
|
||||
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's
|
||||
* scratchpad slots (PS1 hardware scratchpad at 0x1F800000).
|
||||
*
|
||||
* Each slot is 16 bytes: V3_S4 is already 16 bytes (4 × S4 = x/y/z/pad).
|
||||
* The struct fields are contiguous — slot i starts at offset i*16.
|
||||
* Used by the assembly via O_(ResolveLookAtScratch, fld.x/y/z) which resolves
|
||||
* to a compile-time byte offset. NOT a runtime struct — the struct is purely
|
||||
* a schema for offsets; the assembly uses `r_scratch + O_(...)` to compute
|
||||
* slot addresses at runtime.
|
||||
*
|
||||
* Slot producers/consumers (referenced by the resolve_look_at chain atoms):
|
||||
*
|
||||
* +0 fwd atom 0 writes (target - eye); atom 1 (normalize) reads
|
||||
* +16 uz atom 1 writes (normalize fwd); atoms 2 + 4 read (cross operands)
|
||||
* +32 right atom 2 writes (cross uz x up_in); atom 3 (normalize) reads
|
||||
* +48 ux atom 3 writes (normalize right); atoms 4 + 6 read
|
||||
* +64 up atom 4 writes (cross uz x ux); atom 5 (normalize) reads
|
||||
* +80 uy atom 5 writes (normalize up); atom 6 reads
|
||||
* +96 eye atom 0 stages (C-side input); atom 6 reads (translation column)
|
||||
* +112 target reserved (currently written nowhere — kept for symmetry w/ eye)
|
||||
* +128 up_in atom 0 stages (C-side input); atom 2 reads (cross operand)
|
||||
*
|
||||
* Fields use P3_S4 (point) for eye/target (RGA: affine point, implicit weight 1);
|
||||
* V3_S4 (vector) for fwd/uz/right/ux/up/uy/up_in (RGA: Euclidean vector). P3_S4
|
||||
* is a storage alias of V3_S4 (see math.h comment: "Storage alias of V3_S4.
|
||||
* Use P3_S4 when the value is a point.") — both are 16 bytes.
|
||||
*
|
||||
* Moved from gte.atom.c (Task 12.11): gte.atom.c is the GENERIC GTE primitives
|
||||
* file and must not know about any specific atom bundle's scratch layout. */
|
||||
typedef Struct_(ResolveLookAtScratch) {
|
||||
V3_S4 fwd; /* offset +0 (16 bytes — 4 S4 fields incl. internal pad) */
|
||||
V3_S4 uz; /* offset +16 (16 bytes) */
|
||||
V3_S4 right; /* offset +32 (16 bytes) */
|
||||
V3_S4 ux; /* offset +48 (16 bytes) */
|
||||
V3_S4 up; /* offset +64 (16 bytes) */
|
||||
V3_S4 uy; /* offset +80 (16 bytes) */
|
||||
P3_S4 eye; /* offset +96 (16 bytes; storage alias of V3_S4) */
|
||||
P3_S4 target; /* offset +112 (16 bytes; storage alias of V3_S4) */
|
||||
V3_S4 up_in; /* offset +128 (16 bytes) */
|
||||
};
|
||||
|
||||
/* ─── resolve_look_at bundle chain atoms (Task 5) ────────────────────────────
|
||||
* 7 unique atom procs in the resolve_look_at bundle (4 chain atoms + 3 normalize
|
||||
* variants). All 7 are runtime-built MipsAtom_Proc_ atoms: each function declares
|
||||
* a static MipsCode[] body, then calls atombuilder_unroll() to append it to the
|
||||
* caller's MipsAtomBuilder arena. Task 6's resolve_look_at_init() uses this pattern
|
||||
* to pre-build the bundle into the static arena (smem.resolve_look_at_arena).
|
||||
*
|
||||
* Atom roster (positions 0-6 in the bundle):
|
||||
* Atom 0: resolve_look_at__input_and_sub (chain atom)
|
||||
* Atom 1: resolve_look_at__normalize_fwd_to_uz (normalize wrapper)
|
||||
* Atom 2: resolve_look_at__cross_uz_up_in_to_right (chain atom)
|
||||
* Atom 3: resolve_look_at__normalize_right_to_ux (normalize wrapper)
|
||||
* Atom 4: resolve_look_at__cross_uz_ux_to_up (chain atom)
|
||||
* Atom 5: resolve_look_at__normalize_up_to_uy (normalize wrapper)
|
||||
* Atom 6: resolve_look_at__populate_and_translate (chain atom)
|
||||
*
|
||||
* The 3 normalize wrappers are CHAIN-SPECIFIC — they hardcode src/dst scratch
|
||||
* offsets in the body (computed via r_scratch + O_(ResolveLookAtScratch, fld)).
|
||||
* The generic normalize_v3s4_proc (in gte.atom.c) takes src/dst as GPR parameters
|
||||
* and is NOT used by this bundle. (Layering rule: gte.atom.c contains only
|
||||
* generic GTE primitives; bundle-specific code lives in this file.)
|
||||
*
|
||||
* The 3 normalize procs were moved from gte.atom.c to this file in Task 12.11
|
||||
* (user feedback: "normalize is not supposed to be aware of a specific scratch
|
||||
* for one atom bundle"). The procs were renamed to resolve_look_at__normalize_*_proc
|
||||
* to make their bundle-specific nature clear.
|
||||
*
|
||||
* Lua metaprogram support (Task 12.10): the metaprogram auto-emits
|
||||
* `atom_offset__X__Y` defs in gen/offsets.h for each atom_label/atom_offset pair
|
||||
* in the body. The 3 normalize procs each have internal branches (srav_path /
|
||||
* aligned_done variants) and get their per-proc-instance defs (e.g.,
|
||||
* `atom_offset_srav_path_fwd_to_uz_aligned_done_fwd_to_uz`).
|
||||
*/
|
||||
|
||||
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||
U4 target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
|
||||
U4 eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||
U4 up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||
};
|
||||
|
||||
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad
|
||||
* and computes fwd = target - eye.
|
||||
*
|
||||
* Inputs (C-side pointers popped from the tape; NOT scratchpad addresses):
|
||||
* r_target_ptr : P3_S4* (C-side struct; atom 0 reads target.x/y/z directly)
|
||||
* r_eye_ptr : P3_S4* (C-side struct; staged into scratchpad at +96/+100/+104)
|
||||
* r_up_in_ptr : V3_S4* (C-side struct; staged into scratchpad at +128/+132/+136)
|
||||
*
|
||||
* Wave-context output:
|
||||
* r_scratch : R_ResolveScratch (R_T4) — scratch base, read by atoms 1-6
|
||||
*
|
||||
* Bind-pop layout:
|
||||
* Binds_ResolveLookAtSub = 12 bytes (target + eye + up_in ptrs)
|
||||
* Binds_ResolveLookAtScratch = 4 bytes (scratch_base)
|
||||
*
|
||||
* Staging work:
|
||||
* * Stage eye.x/y/z → scratch+96/+100/+104 (for atom 6's translation column)
|
||||
* * Stage up_in.x/y/z → scratch+128/+132/+136 (for atom 2's outer-product operand)
|
||||
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
|
||||
*
|
||||
* GPR codes (assigned by resolve_look_at_init):
|
||||
* r_target_ptr : R_T0
|
||||
* r_eye_ptr : R_T1
|
||||
* r_up_in_ptr : R_T2
|
||||
* r_scratch : R_T4 (R_ResolveScratch; wave-context carrier)
|
||||
* r_tmp0 : R_T3 (stage eye/up_in + load eye.y)
|
||||
* r_tmp1 : R_T5 (stage eye/up_in + load eye.z)
|
||||
* r_tmp2 : R_T6 (stage eye/up_in + load target.x)
|
||||
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
|
||||
* R_AT : hardcoded (load eye.y / eye.z / target.z)
|
||||
* R_V0 : hardcoded (load eye.z / target.z)
|
||||
*
|
||||
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
|
||||
*/
|
||||
I_ void resolve_look_at__input_and_sub_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_target_ptr
|
||||
, U4 r_eye_ptr
|
||||
, U4 r_up_in_ptr
|
||||
, U4 r_scratch
|
||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
|
||||
) MipsAtom_Proc_(resolve_look_at__input_and_sub, ab, {
|
||||
/* Pop the 3 C-side pointers + scratch_base from the tape. */
|
||||
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtScratch,scratch_base)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtScratch)),
|
||||
|
||||
/* Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation
|
||||
* column). Reuse r_tmp0/r_tmp1/r_tmp2. Offsets via O_(ResolveLookAtScratch,*). */
|
||||
load_word(r_tmp0, r_eye_ptr, O_(P3_S4,x)),
|
||||
load_word(r_tmp1, r_eye_ptr, O_(P3_S4,y)),
|
||||
load_word(r_tmp2, r_eye_ptr, O_(P3_S4,z)),
|
||||
nop, /* load-delay */
|
||||
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,eye.x)),
|
||||
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,eye.y)),
|
||||
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye.z)),
|
||||
|
||||
/* Stage up_in.x/y/z into the scratchpad (atom 2 reads these for the outer
|
||||
* product with uz). Reuse r_tmp0/r_tmp1/r_tmp2. */
|
||||
load_word(r_tmp0, r_up_in_ptr, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_up_in_ptr, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_up_in_ptr, O_(V3_S4,z)),
|
||||
nop, /* load-delay */
|
||||
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,up_in.x)),
|
||||
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,up_in.y)),
|
||||
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in.z)),
|
||||
|
||||
/* Compute fwd = target - eye. */
|
||||
load_word(r_tmp0, r_target_ptr, O_(P3_S4,x)),
|
||||
load_word(r_tmp1, r_target_ptr, O_(P3_S4,y)),
|
||||
load_word(r_tmp2, r_target_ptr, O_(P3_S4,z)),
|
||||
load_word(r_tmp3, r_eye_ptr, O_(P3_S4,x)),
|
||||
load_word(R_AT, r_eye_ptr, O_(P3_S4,y)),
|
||||
load_word(R_V0, r_eye_ptr, O_(P3_S4,z)),
|
||||
nop, /* load-delay */
|
||||
sub_u(r_tmp0, r_tmp0, r_tmp3),
|
||||
sub_u(r_tmp1, r_tmp1, R_AT),
|
||||
sub_u(r_tmp2, r_tmp2, R_V0),
|
||||
|
||||
/* Store fwd.x/y/z (atom 1 reads these as the normalize src). */
|
||||
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,fwd.x)),
|
||||
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,fwd.y)),
|
||||
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd.z)),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
/* Atoms 2 + 4 in the bundle: out = a × b (GTE outer product on IR/D vectors).
|
||||
* No bind pop — the three operand pointers (a, b, out) are derived in-body
|
||||
* from r_scratch + hardcoded_offset. Each atom has its own variant because
|
||||
* the offsets are baked into the body and each atom uses unique GPRs.
|
||||
*
|
||||
* GTE register layout (per PSX-SPX + duffle gte.h):
|
||||
* IR1/2/3 = a.x/y/z (mtc2)
|
||||
* VXY0 = b.x (mtc2)
|
||||
* VZ0 = b.y (mtc2)
|
||||
* VXY1 = b.z (mtc2)
|
||||
* OP = outer product
|
||||
* MAC1/2/3 = out.x/y/z (mfc2)
|
||||
*
|
||||
* Pool cost: r_scratch (R_T4 carrier) + 7 body GPRs + R_AT + R_V0 (hardcoded) = 10 GPRs.
|
||||
*/
|
||||
|
||||
/* Atom 2: cross uz × up_in → right. */
|
||||
I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_scratch
|
||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||
, U4 r_d /* load b.x */
|
||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
|
||||
) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, ab, {
|
||||
/* Compute the three scratch pointers from r_scratch. */
|
||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
|
||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
|
||||
nop,
|
||||
|
||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||
load_word(r_a, r_g, O_(V3_S4,x)),
|
||||
load_word(r_b, r_g, O_(V3_S4,y)),
|
||||
load_word(r_c, r_g, O_(V3_S4,z)),
|
||||
nop,
|
||||
|
||||
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (hardcoded; reusing the
|
||||
* body's last two loads is fine because the load-delay slot is the nop
|
||||
* after the third load, and mtc2 below doesn't read these regs). */
|
||||
load_word(r_d, r_h, O_(V3_S4,x)),
|
||||
load_word(R_AT, r_h, O_(V3_S4,y)),
|
||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||
nop,
|
||||
|
||||
/* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */
|
||||
gte_mv_to_data_r(r_a, C2_IR1),
|
||||
gte_mv_to_data_r(r_b, C2_IR2),
|
||||
gte_mv_to_data_r(r_c, C2_IR3),
|
||||
gte_mv_to_data_r(r_d, C2_VXY0), /* D1 = b.x */
|
||||
gte_mv_to_data_r(R_AT, C2_VZ0), /* D2 = b.y */
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1), /* D3 = b.z */
|
||||
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
|
||||
gte_cmdw_outer_product, /* OP fires; MAC1/2/3 = a × b */
|
||||
|
||||
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
|
||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||
gte_mv_from_data_r(r_b, C2_MAC2),
|
||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||
nop, /* MFC2 retirement */
|
||||
|
||||
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
|
||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||
store_word(r_c, r_f, O_(V3_S4,z)),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
/* Atom 4: cross uz × ux → up. */
|
||||
I_ void resolve_look_at__cross_uz_ux_to_up_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_scratch
|
||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||
, U4 r_d /* load b.x */
|
||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
|
||||
) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, ab, {
|
||||
/* Compute the three scratch pointers from r_scratch. */
|
||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
|
||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
|
||||
nop,
|
||||
|
||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||
load_word(r_a, r_g, O_(V3_S4,x)),
|
||||
load_word(r_b, r_g, O_(V3_S4,y)),
|
||||
load_word(r_c, r_g, O_(V3_S4,z)),
|
||||
nop,
|
||||
|
||||
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
|
||||
load_word(r_d, r_h, O_(V3_S4,x)),
|
||||
load_word(R_AT, r_h, O_(V3_S4,y)),
|
||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||
nop,
|
||||
|
||||
/* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */
|
||||
gte_mv_to_data_r(r_a, C2_IR1),
|
||||
gte_mv_to_data_r(r_b, C2_IR2),
|
||||
gte_mv_to_data_r(r_c, C2_IR3),
|
||||
gte_mv_to_data_r(r_d, C2_VXY0),
|
||||
gte_mv_to_data_r(R_AT, C2_VZ0),
|
||||
gte_mv_to_data_r(R_V0, C2_VXY1),
|
||||
nop2,
|
||||
|
||||
gte_cmdw_outer_product,
|
||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||
gte_mv_from_data_r(r_b, C2_MAC2),
|
||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||
nop,
|
||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||
store_word(r_c, r_f, O_(V3_S4,z)),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
/* Atoms 1, 3, 5 in the bundle: chain-specific normalize wrappers around the
|
||||
* generic normalize_v3s4_proc (gte.atom.c). The generic proc takes src/dst as
|
||||
* GPR parameters; these wrappers HARDCODE src/dst via r_scratch + O_(ResolveLookAtScratch, fld)
|
||||
* so the C-side bundle helper doesn't need to push scratchpad addresses via
|
||||
* tb_data between atoms. (Task 12.8 fix: eliminate magic offsets.)
|
||||
*
|
||||
* The 4-stage normalize body (SQR → mfc2 → LZCS → GPF → srav) is identical to
|
||||
* the generic version (GPR-renamed); cycle counts match. The only per-atom
|
||||
* difference is the (src, dst) scratch offsets and the per-proc atom_label
|
||||
* suffixes (srav_path_fwd_to_uz, srav_path_right_to_ux, srav_path_up_to_uy) so
|
||||
* the per-proc-instance offsets are emitted disjointly in gen/offsets.h.
|
||||
*
|
||||
* GPR pool (10 free regs: R_T0..R_T3, R_T5..R_T7, R_V0, R_V1, R_AT; R_T4 reserved for R_ResolveScratch):
|
||||
* r_a : src ptr (overlaps with r_recip_est carrier after the 3 src-loads)
|
||||
* r_b : dst ptr (saved throughout)
|
||||
* r_e/r_f/r_i : src.x/y/z → result.x/y/z (preserved across stages 1-2 via r_d/r_g/r_recip_est scratch)
|
||||
* r_d/r_g : MAC1/2 scratch (dead after stage 2)
|
||||
* r_h : LZCR (saved across stages 3-4)
|
||||
* r_recip_est : |v|² accumulator + sqrtbl[index] + 1/|v| (saved throughout)
|
||||
* r_shift : final srav amount (saved across stages 3-4)
|
||||
*
|
||||
* The Lua metaprogram (Task 12.10) auto-emits:
|
||||
* - `mac_resolve_look_at__normalize_<from>_to_<to>` alias in gen/macs.h
|
||||
* - `atom_offset__srav_path_<from>_to_<to>__aligned_done_<from>_to_<to>` defs in gen/offsets.h
|
||||
*/
|
||||
|
||||
/* Atom 1: normalize fwd (scratch+0) → uz (scratch+16). */
|
||||
I_ void resolve_look_at__normalize_fwd_to_uz_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_scratch
|
||||
, U4 r_a, U4 r_b /* src/dst scratch pointers */
|
||||
, U4 r_e, U4 r_f, U4 r_i /* src.x/y/z → result.x/y/z */
|
||||
, U4 r_d, U4 r_g /* MAC1/2 scratch (dead after stage 2) */
|
||||
, U4 r_h /* LZCR */
|
||||
, U4 r_recip_est
|
||||
, U4 r_shift
|
||||
) MipsAtom_Proc_(resolve_look_at__normalize_fwd_to_uz, ab, {
|
||||
/* Compute src/dst pointers from r_scratch. */
|
||||
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,fwd)), /* r_a = &fwd */
|
||||
add_si(r_b, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_b = &uz */
|
||||
nop,
|
||||
|
||||
/* Load src.x/y/z from r_a into r_e/r_f/r_i. */
|
||||
load_word(r_e, r_a, O_(V3_S4,x)),
|
||||
load_word(r_f, r_a, O_(V3_S4,y)),
|
||||
load_word(r_i, r_a, O_(V3_S4,z)),
|
||||
nop, /* load-delay */
|
||||
|
||||
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
||||
gte_mv_to_data_r(r_e, C2_IR1),
|
||||
gte_mv_to_data_r(r_f, C2_IR2),
|
||||
gte_mv_to_data_r(r_i, C2_IR3),
|
||||
nop, gte_cmdw_sqr,
|
||||
|
||||
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
|
||||
gte_mv_from_data_r(r_d, C2_MAC1),
|
||||
gte_mv_from_data_r(r_g, C2_MAC2),
|
||||
gte_mv_from_data_r(r_recip_est, C2_MAC3),
|
||||
nop,
|
||||
add_u(r_recip_est, r_recip_est, r_g),
|
||||
add_u(r_recip_est, r_recip_est, r_d),
|
||||
gte_mv_to_data_r(r_recip_est, C2_LZCS),
|
||||
nop2,
|
||||
gte_mv_from_data_r(r_h, C2_LZCR),
|
||||
nop,
|
||||
|
||||
/* Stage 3: compute shift amount, align |v|² to bit 24. */
|
||||
and_i( r_h, r_h, -2),
|
||||
li_s( r_shift, 31),
|
||||
sub_s( r_shift, r_shift, r_h),
|
||||
shift_aright(r_shift, r_shift, 1),
|
||||
add_si( r_a, r_h, -24), /* r_a = LZCR - 24 (overlapping with r_recip_est; src ptr no longer needed) */
|
||||
branch_lt_zero(r_a, atom_offset(srav_path_fwd_to_uz, aligned_done_fwd_to_uz)), nop,
|
||||
jump_rel(atom_offset(aligned_done_fwd_to_uz, srav_path_fwd_to_uz)),
|
||||
shift_lleft_var(r_recip_est, r_recip_est, r_a),
|
||||
atom_label(srav_path_fwd_to_uz)
|
||||
li_s( r_a, 24),
|
||||
sub_s( r_a, r_a, r_h),
|
||||
shift_aright_var(r_recip_est, r_recip_est, r_a),
|
||||
atom_label(aligned_done_fwd_to_uz)
|
||||
/* r_recip_est holds |v|² aligned to bit 24. */
|
||||
add_si( r_recip_est, r_recip_est, -64),
|
||||
shift_lleft(r_recip_est, r_recip_est, 1),
|
||||
load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)),
|
||||
or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)),
|
||||
add_u(r_a, r_a, r_recip_est),
|
||||
load_half(r_recip_est, r_a, 0),
|
||||
nop,
|
||||
|
||||
/* Stage 4: GPF + srav finalize. */
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||
gte_mv_to_data_r(r_e, C2_IR1),
|
||||
gte_mv_to_data_r(r_f, C2_IR2),
|
||||
gte_mv_to_data_r(r_i, C2_IR3),
|
||||
nop2,
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(r_e, C2_MAC1),
|
||||
gte_mv_from_data_r(r_f, C2_MAC2),
|
||||
gte_mv_from_data_r(r_i, C2_MAC3),
|
||||
shift_aright_var(r_e, r_e, r_shift),
|
||||
shift_aright_var(r_f, r_f, r_shift),
|
||||
shift_aright_var(r_i, r_i, r_shift),
|
||||
|
||||
/* Store result.x/y/z to r_b (dst ptr = scratch+16). */
|
||||
store_word(r_e, r_b, O_(V3_S4,x)),
|
||||
store_word(r_f, r_b, O_(V3_S4,y)),
|
||||
store_word(r_i, r_b, O_(V3_S4,z)),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
/* Atom 3: normalize right (scratch+32) → ux (scratch+48). */
|
||||
I_ void resolve_look_at__normalize_right_to_ux_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_scratch
|
||||
, U4 r_a, U4 r_b
|
||||
, U4 r_e, U4 r_f, U4 r_i
|
||||
, U4 r_d, U4 r_g
|
||||
, U4 r_h
|
||||
, U4 r_recip_est
|
||||
, U4 r_shift
|
||||
) MipsAtom_Proc_(resolve_look_at__normalize_right_to_ux, ab, {
|
||||
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,right)), /* r_a = &right */
|
||||
add_si(r_b, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_b = &ux */
|
||||
nop,
|
||||
load_word(r_e, r_a, O_(V3_S4,x)),
|
||||
load_word(r_f, r_a, O_(V3_S4,y)),
|
||||
load_word(r_i, r_a, O_(V3_S4,z)),
|
||||
nop,
|
||||
gte_mv_to_data_r(r_e, C2_IR1),
|
||||
gte_mv_to_data_r(r_f, C2_IR2),
|
||||
gte_mv_to_data_r(r_i, C2_IR3),
|
||||
nop, gte_cmdw_sqr,
|
||||
gte_mv_from_data_r(r_d, C2_MAC1),
|
||||
gte_mv_from_data_r(r_g, C2_MAC2),
|
||||
gte_mv_from_data_r(r_recip_est, C2_MAC3),
|
||||
nop,
|
||||
add_u(r_recip_est, r_recip_est, r_g),
|
||||
add_u(r_recip_est, r_recip_est, r_d),
|
||||
gte_mv_to_data_r(r_recip_est, C2_LZCS),
|
||||
nop2,
|
||||
gte_mv_from_data_r(r_h, C2_LZCR),
|
||||
nop,
|
||||
and_i( r_h, r_h, -2),
|
||||
li_s( r_shift, 31),
|
||||
sub_s( r_shift, r_shift, r_h),
|
||||
shift_aright(r_shift, r_shift, 1),
|
||||
add_si( r_a, r_h, -24),
|
||||
branch_lt_zero(r_a, atom_offset(srav_path_right_to_ux, aligned_done_right_to_ux)), nop,
|
||||
jump_rel(atom_offset(aligned_done_right_to_ux, srav_path_right_to_ux)),
|
||||
shift_lleft_var(r_recip_est, r_recip_est, r_a),
|
||||
atom_label(srav_path_right_to_ux)
|
||||
li_s( r_a, 24),
|
||||
sub_s( r_a, r_a, r_h),
|
||||
shift_aright_var(r_recip_est, r_recip_est, r_a),
|
||||
atom_label(aligned_done_right_to_ux)
|
||||
add_si( r_recip_est, r_recip_est, -64),
|
||||
shift_lleft(r_recip_est, r_recip_est, 1),
|
||||
load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)),
|
||||
or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)),
|
||||
add_u(r_a, r_a, r_recip_est),
|
||||
load_half(r_recip_est, r_a, 0),
|
||||
nop,
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||
gte_mv_to_data_r(r_e, C2_IR1),
|
||||
gte_mv_to_data_r(r_f, C2_IR2),
|
||||
gte_mv_to_data_r(r_i, C2_IR3),
|
||||
nop2,
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(r_e, C2_MAC1),
|
||||
gte_mv_from_data_r(r_f, C2_MAC2),
|
||||
gte_mv_from_data_r(r_i, C2_MAC3),
|
||||
shift_aright_var(r_e, r_e, r_shift),
|
||||
shift_aright_var(r_f, r_f, r_shift),
|
||||
shift_aright_var(r_i, r_i, r_shift),
|
||||
store_word(r_e, r_b, O_(V3_S4,x)),
|
||||
store_word(r_f, r_b, O_(V3_S4,y)),
|
||||
store_word(r_i, r_b, O_(V3_S4,z)),
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
/* Atom 5: normalize up (scratch+64) → uy (scratch+80). */
|
||||
I_ void resolve_look_at__normalize_up_to_uy_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_scratch
|
||||
, U4 r_a, U4 r_b
|
||||
, U4 r_e, U4 r_f, U4 r_i
|
||||
, U4 r_d, U4 r_g
|
||||
, U4 r_h
|
||||
, U4 r_recip_est
|
||||
, U4 r_shift
|
||||
) MipsAtom_Proc_(resolve_look_at__normalize_up_to_uy, ab, {
|
||||
add_si(r_a, r_scratch, O_(ResolveLookAtScratch,up)), /* r_a = &up */
|
||||
add_si(r_b, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_b = &uy */
|
||||
nop,
|
||||
load_word(r_e, r_a, O_(V3_S4,x)),
|
||||
load_word(r_f, r_a, O_(V3_S4,y)),
|
||||
load_word(r_i, r_a, O_(V3_S4,z)),
|
||||
nop,
|
||||
gte_mv_to_data_r(r_e, C2_IR1),
|
||||
gte_mv_to_data_r(r_f, C2_IR2),
|
||||
gte_mv_to_data_r(r_i, C2_IR3),
|
||||
nop, gte_cmdw_sqr,
|
||||
gte_mv_from_data_r(r_d, C2_MAC1),
|
||||
gte_mv_from_data_r(r_g, C2_MAC2),
|
||||
gte_mv_from_data_r(r_recip_est, C2_MAC3),
|
||||
nop,
|
||||
add_u(r_recip_est, r_recip_est, r_g),
|
||||
add_u(r_recip_est, r_recip_est, r_d),
|
||||
gte_mv_to_data_r(r_recip_est, C2_LZCS),
|
||||
nop2,
|
||||
gte_mv_from_data_r(r_h, C2_LZCR),
|
||||
nop,
|
||||
and_i( r_h, r_h, -2),
|
||||
li_s( r_shift, 31),
|
||||
sub_s( r_shift, r_shift, r_h),
|
||||
shift_aright(r_shift, r_shift, 1),
|
||||
add_si( r_a, r_h, -24),
|
||||
branch_lt_zero(r_a, atom_offset(srav_path_up_to_uy, aligned_done_up_to_uy)), nop,
|
||||
jump_rel(atom_offset(aligned_done_up_to_uy, srav_path_up_to_uy)),
|
||||
shift_lleft_var(r_recip_est, r_recip_est, r_a),
|
||||
atom_label(srav_path_up_to_uy)
|
||||
li_s( r_a, 24),
|
||||
sub_s( r_a, r_a, r_h),
|
||||
shift_aright_var(r_recip_est, r_recip_est, r_a),
|
||||
atom_label(aligned_done_up_to_uy)
|
||||
add_si( r_recip_est, r_recip_est, -64),
|
||||
shift_lleft(r_recip_est, r_recip_est, 1),
|
||||
load_upper_i(r_a, u4_hi(& gte_normalize_sqr_tbl)),
|
||||
or_i_self(r_a, u4_lo(& gte_normalize_sqr_tbl)),
|
||||
add_u(r_a, r_a, r_recip_est),
|
||||
load_half(r_recip_est, r_a, 0),
|
||||
nop,
|
||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||
gte_mv_to_data_r(r_e, C2_IR1),
|
||||
gte_mv_to_data_r(r_f, C2_IR2),
|
||||
gte_mv_to_data_r(r_i, C2_IR3),
|
||||
nop2,
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(r_e, C2_MAC1),
|
||||
gte_mv_from_data_r(r_f, C2_MAC2),
|
||||
gte_mv_from_data_r(r_i, C2_MAC3),
|
||||
shift_aright_var(r_e, r_e, r_shift),
|
||||
shift_aright_var(r_f, r_f, r_shift),
|
||||
shift_aright_var(r_i, r_i, r_shift),
|
||||
store_word(r_e, r_b, O_(V3_S4,x)),
|
||||
store_word(r_f, r_b, O_(V3_S4,y)),
|
||||
store_word(r_i, r_b, O_(V3_S4,z)),
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
|
||||
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
|
||||
};
|
||||
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute
|
||||
* the translation column t[] = R * (-eye).
|
||||
*
|
||||
* GPR codes (assigned by resolve_look_at_init):
|
||||
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
||||
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
|
||||
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
|
||||
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
|
||||
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
|
||||
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
|
||||
*
|
||||
* The 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
|
||||
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
|
||||
*
|
||||
* Struct layout (per duffle/math.h):
|
||||
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
||||
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
|
||||
*
|
||||
* Translation column: GTE MVMVA with the world rotation matrix pre-set (helper emits set_gte_world before the bundle, per the bundle design).
|
||||
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
|
||||
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
|
||||
*/
|
||||
I_ void resolve_look_at__populate_and_translate_proc(MipsAtomBuilder_R ab
|
||||
, U4 r_look_at
|
||||
, U4 r_scratch
|
||||
, U4 r_pux, U4 r_puy, U4 r_puz, U4 r_peye /* 4 dedicated pointer regs */
|
||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2 /* 3 atom-local scratch regs */
|
||||
) MipsAtom_Proc_(resolve_look_at__populate_and_translate, ab, {
|
||||
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
|
||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||
|
||||
/* Compute the 4 scratch pointers in their dedicated GPRs. */
|
||||
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
|
||||
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
|
||||
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
|
||||
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)), /* r_peye = &eye */
|
||||
nop,
|
||||
|
||||
/* ── m[0] = (S2)ux ── */
|
||||
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
|
||||
nop,
|
||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
|
||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
|
||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
|
||||
|
||||
/* ── m[1] = (S2)uy ── */
|
||||
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
|
||||
nop,
|
||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
|
||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
|
||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
|
||||
|
||||
/* ── m[2] = (S2)uz ── */
|
||||
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
|
||||
nop,
|
||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
|
||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
|
||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
|
||||
|
||||
/* ── Translation column t[i] = R * (-eye) ─────────────────────────────
|
||||
* pos = -eye: load eye.x/y/z from r_peye, negate via sub_u from R_0. */
|
||||
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
|
||||
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
|
||||
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
|
||||
nop,
|
||||
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
|
||||
sub_u(r_tmp1, R_0, r_tmp1),
|
||||
sub_u(r_tmp2, R_0, r_tmp2),
|
||||
|
||||
/* mtc2 IR1/2/3 = pos (for MVMVA — input vector registers). */
|
||||
gte_mv_to_data_r(r_tmp0, C2_IR1),
|
||||
gte_mv_to_data_r(r_tmp1, C2_IR2),
|
||||
gte_mv_to_data_r(r_tmp2, C2_IR3),
|
||||
nop2,
|
||||
|
||||
/* MVMVA: MAC1/2/3 = R * IR with cv=0 (no TR vector), mx=0 (rotation matrix),
|
||||
* sf=0 (no shift), v=0 (V0 = IR1/2/3, no far-plane clipping). The pre-set
|
||||
* rotation matrix is the one set by the preceding set_gte_world atom.
|
||||
* gte_cmdw_mvmva is parameterless and defaults to cv=0/mx=0/sf=0/v=0. */
|
||||
gte_cmdw_mvmva,
|
||||
nop, /* GTE interlock */
|
||||
|
||||
/* mfc2 MAC1/2/3 → r_tmp0/r_tmp1/r_tmp2 (sign-extended into 32-bit GPRs).
|
||||
* MAC1/2/3 hold R*v with no TR add and no perspective divide — exactly the
|
||||
* 3 distinct world-space translation values we need for t[0..2]. */
|
||||
gte_mv_from_data_r(r_tmp0, C2_MAC1),
|
||||
gte_mv_from_data_r(r_tmp1, C2_MAC2),
|
||||
gte_mv_from_data_r(r_tmp2, C2_MAC3),
|
||||
nop,
|
||||
store_word(r_tmp0, r_look_at, O_(MT3_S2S4,t[0])),
|
||||
store_word(r_tmp1, r_look_at, O_(MT3_S2S4,t[1])),
|
||||
store_word(r_tmp2, r_look_at, O_(MT3_S2S4,t[2])),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
#pragma endregion Atom Procs
|
||||
|
||||
#pragma region Baked Atoms
|
||||
|
||||
enum {
|
||||
@@ -180,37 +850,17 @@ internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads
|
||||
mac_yield(),
|
||||
};
|
||||
|
||||
/* ----- pad_apply_input -----
|
||||
* Reads pad[0].buttons + pad[0].left_x;
|
||||
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
|
||||
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
|
||||
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
|
||||
* - Analog stick X (dead zone 0x70..0x90):
|
||||
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
|
||||
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
|
||||
* - D-pad + analog deltas add when used together.
|
||||
*
|
||||
* Convention:
|
||||
* pad_state = 0 means no buttons active.
|
||||
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
|
||||
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
|
||||
*
|
||||
* Signed-delta trick:
|
||||
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
|
||||
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
|
||||
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
|
||||
*/
|
||||
typedef Struct_(Binds_PadApplyInput) {
|
||||
PadState* state;
|
||||
V3_S2* cube_rot;
|
||||
V3_S2* floor_rot;
|
||||
PadState* state;
|
||||
V3_S2* cube_rot;
|
||||
V3_S2* floor_rot;
|
||||
};
|
||||
enum {
|
||||
R_PadStateT5 = R_T5 atom_reg,
|
||||
R_CubeRot = R_T1 atom_reg,
|
||||
R_FloorRot = R_T2 atom_reg,
|
||||
};
|
||||
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||
, atom_writes( R_CubeRot, R_FloorRot)
|
||||
) {
|
||||
@@ -225,7 +875,7 @@ internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
// Note(Ed): Potential op with delay slot?
|
||||
|
||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, 30),
|
||||
@@ -235,7 +885,7 @@ internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
atom_label(exit_dpad_left)
|
||||
|
||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, -30),
|
||||
@@ -246,21 +896,21 @@ internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
|
||||
|
||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)),
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
|
||||
|
||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||
|
||||
atom_label(dead_check_upper)
|
||||
/* left_x >= 0x70 → check upper bound. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */
|
||||
add_ui( R_T4, R_0, 0x90),
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
|
||||
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
||||
|
||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||
mac_yield_load(),
|
||||
|
||||
@@ -273,8 +923,7 @@ atom_label(dead_low_active)
|
||||
|
||||
/* R_T4 = cube_delta */
|
||||
shift_aright(R_T4, R_T3, 2),
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
nop,
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||
@@ -295,8 +944,7 @@ atom_label(dead_high_active)
|
||||
/* delta = 0x80 - left_x (signed negative). */
|
||||
|
||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
nop,
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
|
||||
@@ -314,6 +962,94 @@ atom_label(exit_stick)
|
||||
mac_yield_tail(),
|
||||
};
|
||||
|
||||
enum {
|
||||
R_Cam = R_T4 atom_reg,
|
||||
R_CamPadState = R_T5 atom_reg,
|
||||
};
|
||||
typedef Struct_(Binds_PadInputCam) {
|
||||
PadState* state;
|
||||
Camera* cam;
|
||||
};
|
||||
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
||||
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
|
||||
, atom_writes(R_Cam)
|
||||
) {
|
||||
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
||||
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
||||
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
|
||||
|
||||
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
atom_label(exit_left_x)
|
||||
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
|
||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
atom_label(exit_right_x)
|
||||
|
||||
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||
atom_label(exit_up_y)
|
||||
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
||||
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
|
||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||
atom_label(exit_down_y)
|
||||
|
||||
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||
atom_label(exit_cross_z)
|
||||
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
||||
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
|
||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||
atom_label(exit_circle_z)
|
||||
|
||||
mac_yield_tail(),
|
||||
};
|
||||
|
||||
/* Scratchpad layout for the resolve_look_at bundle.
|
||||
* The chain atoms communicate entirely via the wave-context GPR carrier
|
||||
* R_ResolveScratch (R_T4) + hardcoded offsets into smem.scratchpad
|
||||
* (PS1 hardware scratchpad at 0x1F800000).
|
||||
*
|
||||
* Atom 0 (input_and_sub) STAGES the C-side inputs (eye, up_in) into the scratchpad;
|
||||
* AT THE SAME TIME it computes fwd = target - eye and stores it at scratch+0.
|
||||
* Atoms 1-6 then read/write specific scratchpad offsets internally using
|
||||
* `r_scratch + hardcoded_offset` — no tape-data pointers are passed between atoms.
|
||||
*
|
||||
* +0 fwd (atom 0 writes; atom 1 reads)
|
||||
* +16 uz (atom 1 writes; atoms 2 + 4 read)
|
||||
* +32 right (atom 2 writes; atom 3 reads)
|
||||
* +48 ux (atom 3 writes; atoms 4 + 6 read)
|
||||
* +64 up (atom 4 writes; atom 5 reads)
|
||||
* +80 uy (atom 5 writes; atom 6 reads)
|
||||
* +96 eye (atom 0 stages from C-side pointer; atom 6 reads)
|
||||
* +128 up_in (atom 0 stages from C-side pointer; atom 2 reads)
|
||||
*
|
||||
* No struct view is required — the C-side bundle helper passes only C-side
|
||||
* pointers (target, eye, up_in, look_at) and the scratch base address;
|
||||
* the assembly hardcodes all inter-slot offsets. The original Task 12.7 magic
|
||||
* offsets `& smem.scratchpad[N]` in the C-side helper were eliminated by this
|
||||
* redesign; the user feedback was: "you didn't have to use magic offsets into
|
||||
* the scratchpad memory. those are harcoded." */
|
||||
|
||||
enum {
|
||||
R_LookAt = R_T0 atom_reg atom_type(MT3_S2S4*),
|
||||
R_CamEye = R_T1 atom_reg atom_type(P3_S4*),
|
||||
R_CamTarget = R_T2 atom_reg atom_type(P3_S4*),
|
||||
R_WorldUp = R_T3 atom_reg atom_type(V3_S4*),
|
||||
};
|
||||
|
||||
|
||||
|
||||
enum {
|
||||
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||
@@ -344,7 +1080,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||
internal
|
||||
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||
@@ -363,7 +1099,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||
* harmless because the OT entry that points to this prim is created later, only on the body path. */
|
||||
* harmless because the OT entry that points to this prim is created later. */
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
@@ -379,7 +1115,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
|
||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||
mac_format_g4_color(R_PrimCursor,
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||
@@ -420,7 +1156,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices( R_FaceCursor, R_T0, R_T1, R_T2),
|
||||
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
@@ -439,7 +1175,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
|
||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
|
||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||
// end: branch(bounds_chk)
|
||||
|
||||
@@ -26,10 +26,12 @@
|
||||
#include "duffle/dsl.atom.h"
|
||||
#include "duffle/lottes_tape.h"
|
||||
|
||||
#include "duffle/bios.h"
|
||||
#include "duffle/psyq.h"
|
||||
#pragma endregion Duffle Headers
|
||||
|
||||
#pragma region Duffle TUs
|
||||
#include "duffle/pad.c"
|
||||
#include "duffle/math.atom.c"
|
||||
#include "duffle/mips.atom.c"
|
||||
#include "duffle/gte.atom.c"
|
||||
@@ -50,8 +52,9 @@
|
||||
#pragma endregion Hello Joypad TUs
|
||||
|
||||
enum {
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
ResolveLookAtArena_Words = 512,
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
PrimitiveArena primitives;
|
||||
@@ -61,7 +64,10 @@ typedef Struct_(SMemory) {
|
||||
|
||||
U4 MemTape[MemTape_Len];
|
||||
|
||||
M3_S2 tform_world;
|
||||
MT3_S2S4 tform_world;
|
||||
MT3_S2S4 tform_view;
|
||||
|
||||
Camera cam;
|
||||
|
||||
Ent_Cube cube;
|
||||
Ent_Floor floor;
|
||||
@@ -70,10 +76,24 @@ typedef Struct_(SMemory) {
|
||||
PadState pad[2];
|
||||
|
||||
U4_V scratchpad; // d-cache
|
||||
|
||||
/* resolve_look_at bundle: pre-built atom arena + atom-refs.
|
||||
* (Task 12.5 fix: moved from file-scope globals to smem fields.
|
||||
* Task 12.7 fix: dropped the ResolveLookAtScratch struct-as-view; the
|
||||
* C-side helper uses `& smem.scratchpad[N]` at hardcoded offsets directly.
|
||||
* Task 12.8 fix: chain atoms use r_scratch + offset internally; no C-side magic offsets anywhere.
|
||||
* Task 12.11 fix: ResolveLookAtScratch offset schema moved to hello_camera.atom.c — gte.atom.c
|
||||
* is the GENERIC GTE primitives file and must not know about the resolve_look_at bundle's scratch layout.) */
|
||||
U4 resolve_look_at_arena[ResolveLookAtArena_Words]; /* ~2 KB; bumped from 420 per Task 4 subagent */
|
||||
MipsAtom* resolve_look_at_atom_addrs[7];
|
||||
MipsAtomBuilder resolve_look_at_ab_static;
|
||||
};
|
||||
global SMemory smem;
|
||||
extern SMemory smem;
|
||||
|
||||
#define pad0_btn_(btn) btn & smem.pad[0].buttons
|
||||
#define pad1_btn_(btn) btn & smem.pad[1].buttons
|
||||
|
||||
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
gknown PrimitiveArena* pa = & smem.primitives;
|
||||
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||
@@ -84,97 +104,212 @@ I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||
}
|
||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||
|
||||
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated.
|
||||
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||
*
|
||||
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly).
|
||||
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||
*
|
||||
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
|
||||
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||
{
|
||||
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||
(void)p0; (void)p1;
|
||||
void
|
||||
resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
||||
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
||||
V3_S4 right, up, forward;
|
||||
V3_S4 ux, uy, uz;
|
||||
V3_S4 pos, off;
|
||||
|
||||
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||
// Use enums.
|
||||
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
|
||||
normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
||||
|
||||
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||
* $a3 = 0x22 (immediate)
|
||||
* $t1 = 0x12 (function number)
|
||||
* $t2 = 0xB0 (BIOS B-table address) */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
||||
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
|
||||
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
|
||||
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
|
||||
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_rpins, r_use(p0), r_use(p1)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
||||
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
||||
|
||||
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||
u1_v(raw0)[0] = 0xFF;
|
||||
u1_v(raw1)[0] = 0xFF;
|
||||
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
||||
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
|
||||
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
|
||||
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
|
||||
|
||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||
asm volatile(
|
||||
asm_words(
|
||||
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */
|
||||
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */
|
||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
||||
nop /* BD slot */
|
||||
)
|
||||
asm_clobber:
|
||||
rlit(R_AT),
|
||||
rlit(R_V0), rlit(R_V1),
|
||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||
rlit(R_RA),
|
||||
clb_mem_drain
|
||||
);
|
||||
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
|
||||
|
||||
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
|
||||
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
|
||||
mul_m3s2_v3s4(look_at, & pos, & off);
|
||||
trans_m3s2( look_at, & off);
|
||||
}
|
||||
|
||||
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
|
||||
* Called ONCE from main() before the frame loop.
|
||||
* After this returns, the smem.resolve_look_at_atom_addrs[] array contains valid MIPS atom pointers
|
||||
* for the frame-time bundle helper to emit via tb_emit(tb, captured_addr).
|
||||
*
|
||||
* 7 atoms are within hello_camera.atom.c:
|
||||
* 0: resolve_look_at__input_and_sub_proc
|
||||
* 1: resolve_look_at__normalize_fwd_to_uz_proc
|
||||
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
|
||||
* 3: resolve_look_at__normalize_right_to_ux_proc
|
||||
* 4: resolve_look_at__cross_uz_ux_to_up_proc
|
||||
* 5: resolve_look_at__normalize_up_to_uy_proc
|
||||
* 6: resolve_look_at__populate_and_translate_proc
|
||||
*
|
||||
* (gte.atom.c contains only normalize_v3s4_proc — bundle-specific scratch layout is no longer exposed to the GTE primitives file.)
|
||||
*
|
||||
* GPR pool per atom: 10 free GPRs (R_T0..R_T3 + R_T5..R_T7 + R_V0 + R_V1 + R_AT).
|
||||
* R_T4 is reserved as the wave-context carrier (R_ResolveScratch).
|
||||
*/
|
||||
internal void resolve_look_at_init(void) {
|
||||
/* Wrap the static arena in a MipsAtomBuilder. */
|
||||
MipsAtomBuilder_R ab = & smem.resolve_look_at_ab_static;
|
||||
ab->start = u4_(smem.resolve_look_at_arena);
|
||||
ab->capacity = ResolveLookAtArena_Words;
|
||||
ab->used = 0;
|
||||
|
||||
/* Atom 0: resolve_look_at__input_and_sub — stages eye/up_in into scratchpad,
|
||||
* computes fwd = target - eye; binds R_ResolveScratch (R_T4) as the wave-context carrier for atoms 1-6.
|
||||
* The body hardcodes R_AT and R_V0 as eye.y/eye.z temps (the existing sub_u(eye.x, eye.y, eye.z) chain from the prior Task 12.7 design). */
|
||||
smem.resolve_look_at_atom_addrs[0] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__input_and_sub_proc(ab,
|
||||
R_T0, /* r_target_ptr (popped from tape) */
|
||||
R_T1, /* r_eye_ptr (popped from tape) */
|
||||
R_T2, /* r_up_in_ptr (popped from tape) */
|
||||
R_ResolveScratch, /* r_scratch (wave-context carrier; popped from tape) */
|
||||
R_T3, /* r_tmp0 */
|
||||
R_T5, /* r_tmp1 */
|
||||
R_T6, /* r_tmp2 */
|
||||
R_T7); /* r_tmp3 */
|
||||
|
||||
/* Atom 1: resolve_look_at__normalize_fwd_to_uz — src=scratch+0, dst=scratch+16 (HARDCODED in body).
|
||||
* GPR pool: r_scratch (R_T4 carrier) + 10 body GPRs = 11.
|
||||
* r_a/r_b (R_T0/R_T1) : src/dst pointers (r_a overlaps r_recip_est after the loads)
|
||||
* r_e/r_f/r_i (R_T2/R_T3/R_T5) : src.x/y/z → result.x/y/z
|
||||
* r_d/r_g (R_T6/R_T7) : MAC1/2 scratch (dead after stage 2)
|
||||
* r_h (R_V0) : LZCR
|
||||
* r_recip_est (R_V1), r_shift (R_AT) : saved throughout */
|
||||
smem.resolve_look_at_atom_addrs[1] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__normalize_fwd_to_uz_proc(ab,
|
||||
R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */
|
||||
R_T0, R_T1, /* r_a, r_b (src/dst ptrs) */
|
||||
R_T2, R_T3, R_T5, /* r_e, r_f, r_i (src components) */
|
||||
R_T6, R_T7, /* r_d, r_g (MAC scratch) */
|
||||
R_V0, /* r_h (LZCR) */
|
||||
R_V1, /* r_recip_est */
|
||||
R_AT); /* r_shift */
|
||||
|
||||
/* Atom 2: resolve_look_at__cross_uz_up_in_to_right — a=scratch+16, b=scratch+128,
|
||||
* out=scratch+32 (HARDCODED in body). GPR pool: r_scratch + 7 body + R_AT + R_V0 = 10. */
|
||||
smem.resolve_look_at_atom_addrs[2] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__cross_uz_up_in_to_right_proc(ab,
|
||||
R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */
|
||||
R_T0, R_T1, R_T2, /* r_a, r_b, r_c (a.x/y/z → out.x/y/z) */
|
||||
R_T3, /* r_d (b.x) */
|
||||
R_T5, /* r_f (out ptr = scratch+32) */
|
||||
R_T6, /* r_g (a ptr = scratch+16) */
|
||||
R_T7); /* r_h (b ptr = scratch+128) */
|
||||
|
||||
/* Atom 3: resolve_look_at__normalize_right_to_ux — src=scratch+32, dst=scratch+48 (HARDCODED). */
|
||||
smem.resolve_look_at_atom_addrs[3] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__normalize_right_to_ux_proc(ab,
|
||||
R_ResolveScratch,
|
||||
R_T0, R_T1,
|
||||
R_T2, R_T3, R_T5,
|
||||
R_T6, R_T7,
|
||||
R_V0,
|
||||
R_V1,
|
||||
R_AT);
|
||||
|
||||
/* Atom 4: resolve_look_at__cross_uz_ux_to_up — a=scratch+16, b=scratch+48, out=scratch+64 (HARDCODED). */
|
||||
smem.resolve_look_at_atom_addrs[4] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__cross_uz_ux_to_up_proc(ab,
|
||||
R_ResolveScratch,
|
||||
R_T0, R_T1, R_T2,
|
||||
R_T3,
|
||||
R_T5, /* r_f (out ptr = scratch+64) */
|
||||
R_T6, /* r_g (a ptr = scratch+16) */
|
||||
R_T7); /* r_h (b ptr = scratch+48) */
|
||||
|
||||
/* Atom 5: resolve_look_at__normalize_up_to_uy — src=scratch+64, dst=scratch+80 (HARDCODED). */
|
||||
smem.resolve_look_at_atom_addrs[5] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__normalize_up_to_uy_proc(ab,
|
||||
R_ResolveScratch,
|
||||
R_T0, R_T1,
|
||||
R_T2, R_T3, R_T5,
|
||||
R_T6, R_T7,
|
||||
R_V0,
|
||||
R_V1,
|
||||
R_AT);
|
||||
|
||||
/* Atom 6: resolve_look_at__populate_and_translate — write look_at->m[][] from ux/uy/uz (computed from r_scratch+offset internally),
|
||||
then compute translation column t[] = R * (-eye). GPR pool: r_look_at + r_scratch + 4 ptr regs + 3 tmp regs = 9. */
|
||||
smem.resolve_look_at_atom_addrs[6] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
||||
resolve_look_at__populate_and_translate_proc(ab,
|
||||
R_T0, /* r_look_at (popped from tape; MT3_S2S4*) */
|
||||
R_ResolveScratch, /* r_scratch (wave-context carrier) */
|
||||
R_T1, R_T3, R_T5, R_T7, /* r_pux, r_puy, r_puz, r_peye */
|
||||
R_T2, R_T6, R_V0); /* r_tmp0, r_tmp1, r_tmp2 */
|
||||
|
||||
/* Sanity check: arena didn't overflow. */
|
||||
assert(ab->used <= ResolveLookAtArena_Words);
|
||||
}
|
||||
|
||||
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
||||
* The 7 chain atoms are pre-built at init time (resolve_look_at_init) and referenced by address via smem.resolve_look_at_atom_addrs[].
|
||||
* Per-frame work: 7 tb_emit (atom pointer emissions) + 5 tb_data (C-side pointers for atom 0 + look_at for atom 6).
|
||||
*
|
||||
* Binds_ contract (the field-name labels are for human readability):
|
||||
* Atom 0 input_and_sub target(4) eye(4) up_in(4) scratch_base(4) = 4 words
|
||||
* Atoms 1-5 (no tape data — atom uses r_scratch + offset internally)
|
||||
* Atom 6 populate_and_translate look_at(4) = 1 word
|
||||
* ----
|
||||
* 5 tb_data words total per frame.
|
||||
*/
|
||||
I_ void resolve_look_at(
|
||||
TapeBuilder_R tb
|
||||
, MT3_S2S4* look_at
|
||||
, P3_S4* eye
|
||||
, P3_S4* target
|
||||
, V3_S4* up_in
|
||||
){
|
||||
/* Atom 0: input_and_sub — stages eye/up_in into scratchpad + computes fwd. */
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
|
||||
tb_data(tb, u4_(target)); /* Binds_ResolveLookAtSub.target (C-side P3_S4*) */
|
||||
tb_data(tb, u4_(eye)); /* Binds_ResolveLookAtSub.eye (C-side P3_S4*) */
|
||||
tb_data(tb, u4_(up_in)); /* Binds_ResolveLookAtSub.up_in (C-side V3_S4*) */
|
||||
tb_data(tb, u4_(smem.scratchpad)); /* Binds_ResolveLookAtScratch.scratch_base */
|
||||
}
|
||||
|
||||
/* Atoms 1-5: NO tb_data — each chain atom uses r_scratch + hardcoded_offset internally (no tape-data pointers between atoms).
|
||||
Context carrier R_ResolveScratch (R_T4) is preserved across atoms. */
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
|
||||
|
||||
/* Atom 6: populate_and_translate — only output pointer is the matrix destination. */
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
|
||||
tb_data(tb, u4_(look_at)); /* Binds_ResolveLookAtPopAndTrans.look_at (MT3_S2S4*) */
|
||||
}
|
||||
}
|
||||
|
||||
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||
|
||||
GCC_OPTIMIZATION_DISABLE
|
||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
{
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||
|
||||
if (1) // Pad Input
|
||||
// Pad Input
|
||||
{
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
/* BIOS-owned polling: per-frame snapshot of both ports. */
|
||||
// Grab latest state from bios.
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[0]);
|
||||
tb_data_(raw, & smem.pad_raw[0]);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_emit_(pad_bios_snapshot);
|
||||
tb_data_(raw, & smem.pad_raw[1]);
|
||||
tb_data_(state, & smem.pad[1]);
|
||||
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
|
||||
tb_emit_(pad_apply_input);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_data_(cube_rot, & smem.cube.rot);
|
||||
tb_data_(floor_rot, & smem.floor.rot);
|
||||
|
||||
tb_emit_(pad_input_cam);
|
||||
tb_data_(state, & smem.pad[0]);
|
||||
tb_data_(cam, & smem.cam);
|
||||
|
||||
// tb_emit_(pad_input_cube_rotation);
|
||||
// tb_data_(state, & smem.pad[0]);
|
||||
// tb_data_(cube_rot, & smem.cube.rot);
|
||||
// tb_data_(floor_rot, & smem.floor.rot);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -201,15 +336,36 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
A2_S2 p; //???
|
||||
S4 flag; //????
|
||||
|
||||
// Camera Look at
|
||||
if (1)
|
||||
{
|
||||
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||
}
|
||||
// Camera look at (Tape)
|
||||
{
|
||||
MT3_S2S4* look_at = & smem.cam.look_at;
|
||||
P3_S4* eye = & smem.cam.pos;
|
||||
V3_S4* up_in = & v3s4(0, -fp_one, 0);
|
||||
|
||||
tb.used = 0; tb_scope_run(& tb) {
|
||||
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||
}
|
||||
}
|
||||
|
||||
// Draw cube
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.cube.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.cube.scale);
|
||||
gte_matrix_set_rotation (& smem.tform_world);
|
||||
gte_matrix_set_translation(& smem.tform_world);
|
||||
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
|
||||
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
|
||||
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
|
||||
|
||||
// Combine world and look_at matrix.
|
||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||
gte_matrix_set_rotation (& smem.tform_view);
|
||||
gte_matrix_set_translation(& smem.tform_view);
|
||||
|
||||
// gte_matrix_set_rotation (& smem.tform_world);
|
||||
// gte_matrix_set_translation(& smem.tform_world);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
@@ -230,16 +386,22 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));
|
||||
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||
|
||||
// smem.cube.rot.y += 30;
|
||||
}
|
||||
// Draw floor
|
||||
if (1)
|
||||
{
|
||||
m3s2_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
m3s2_translation(& smem.tform_world, & smem.floor.pos);
|
||||
m3s2_scale (& smem.tform_world, & smem.floor.scale);
|
||||
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
|
||||
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
|
||||
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
|
||||
|
||||
// Combine world and look_at matrix.
|
||||
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||
|
||||
gte_matrix_set_rotation (& smem.tform_view);
|
||||
gte_matrix_set_translation(& smem.tform_view);
|
||||
|
||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||
U4 prim_cursor = prim_base + pa->used;
|
||||
@@ -249,11 +411,11 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
|
||||
// Prepare the tape. (Push protocol to tape)
|
||||
tb.used = 0; tb_scope(& tb) {
|
||||
tb_emit(& tb, set_gte_world);
|
||||
tb_data(& tb, u4_(& smem.tform_world));
|
||||
// tb_emit(& tb, set_gte_mt3s2s4);
|
||||
// tb_data(& tb, u4_(& smem.tform_view));
|
||||
|
||||
tb_emit(& tb, rbind_floor_f3_face);
|
||||
// TODO(Ed): Just use a single context struct ref
|
||||
// TODO(Ed): Just use a single context struct ref?
|
||||
tb_data(& tb, prim_cursor);
|
||||
tb_data(& tb, u4_(smem.floor.faces));
|
||||
tb_data(& tb, u4_(smem.floor.verts));
|
||||
@@ -266,7 +428,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||
tb_data(& tb, u4_(& pa->used));
|
||||
tb_data(& tb, prim_base);
|
||||
}
|
||||
tape_run(tb_slice(tb));// Fire off the tape.
|
||||
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||
|
||||
// C-side state (pa->used) has already been updated by the tape!
|
||||
// smem.floor.rot.y += 5;
|
||||
@@ -296,6 +458,7 @@ int main(void)
|
||||
smem.scratchpad = C_(U4_V, 0x1F800000);
|
||||
// smem.primitives.used = 0;
|
||||
// smem.active_buf_id = 0;
|
||||
smem.cam.pos = v3s4(500, -1000, -1500);
|
||||
/*Persistent Entity Setup*/{
|
||||
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||
Ent_Cube* cube = & smem.cube;
|
||||
@@ -315,6 +478,10 @@ int main(void)
|
||||
reset_graph(0);
|
||||
/* Direct BIOS: poll both ports during VBlank. */
|
||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||
|
||||
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
|
||||
resolve_look_at_init();
|
||||
|
||||
/* Pinned registers for the GPU init atom. */
|
||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||
|
||||
@@ -21,12 +21,6 @@ enum {
|
||||
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||
};
|
||||
|
||||
enum {
|
||||
fp_one = (1 << 12),
|
||||
};
|
||||
|
||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||
|
||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||
typedef Array_(OrderingTable_Buffer, 2);
|
||||
|
||||
@@ -67,7 +61,7 @@ I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||
typedef Struct_(Ent_Cube) {
|
||||
V3_S4 accel;
|
||||
V3_S4 vel;
|
||||
V3_S4 pos;
|
||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A8_V3_S2 verts;
|
||||
@@ -94,9 +88,15 @@ I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||
};
|
||||
typedef Struct_(Ent_Floor) {
|
||||
V3_S4 accel;
|
||||
V3_S4 pos;
|
||||
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S4 scale;
|
||||
V3_S2 rot;
|
||||
A4_V3_S2 verts;
|
||||
A2_V3_S2 faces;
|
||||
};
|
||||
|
||||
typedef Struct_(Camera) {
|
||||
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||
V3_S2 rot;
|
||||
MT3_S2S4 look_at;
|
||||
};
|
||||
|
||||
@@ -24,8 +24,8 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||
|
||||
#pragma region MACs (Mips Atom components)
|
||||
|
||||
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
|
||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||
@@ -35,8 +35,8 @@ MipsAtomComp_Proc_(ac_put_disp_env, {
|
||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_draw_env, {
|
||||
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
|
||||
/*
|
||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||
* References:
|
||||
@@ -116,7 +116,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||
|
||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||
|
||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||
|
||||
@@ -286,7 +286,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||
) {
|
||||
mac_load_tri_indices( R_FaceCursor, R_T0, R_T1, R_T2),
|
||||
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||
gte_cmdw_nclip,
|
||||
|
||||
@@ -24,8 +24,8 @@
|
||||
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
||||
* two-instruction zero-extended buttons load).
|
||||
*/
|
||||
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, {
|
||||
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
|
||||
add_ui(scratch_reg, R_0, status_val),
|
||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
||||
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
||||
|
||||
+5
-11
@@ -180,12 +180,9 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
||||
|
||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim.
|
||||
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math,
|
||||
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but
|
||||
# ZERO .o files pulled in — they were unused. The 5 kept libraries
|
||||
# (api, c, etc, gpu, gte) are required by the C-side calls in
|
||||
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
|
||||
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
|
||||
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||
$libraries = @(
|
||||
"api",
|
||||
"c",
|
||||
@@ -227,9 +224,7 @@ function ps1-meta { param(
|
||||
[string[]]$passes = @('--pre-link'),
|
||||
[string[]]$extra_args = @()
|
||||
)
|
||||
# `--unity-root` and `--source` are
|
||||
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
|
||||
# be supplied; the other must be absent.
|
||||
# `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
|
||||
if ($null -ne $unity_root -and $unity_root -ne '')
|
||||
{
|
||||
if ($null -ne $sources -and $sources.Count -gt 0) {
|
||||
@@ -522,7 +517,7 @@ function build-hello_camera {
|
||||
$path_build_gen = join-path $path_build 'gen'
|
||||
|
||||
$src_c = join-path $path_module 'hello_camera.c'
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
|
||||
|
||||
$assemble_args = @()
|
||||
$assemble_args += $f_debug
|
||||
@@ -557,7 +552,6 @@ function build-hello_camera {
|
||||
link-modules $link_modules $elf $link_args
|
||||
make-binary $elf $exe
|
||||
|
||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||
|
||||
inject-dwarf $elf $path_build_gen
|
||||
|
||||
+64
-10
@@ -1053,6 +1053,8 @@ M.GTE_COMMAND_ALIASES = {
|
||||
-- gte_avg_sort_z3 / gte_avg_sort_z4 are the duffle-side aliases for AVSZ3/4.
|
||||
["gte_avg_sort_z3"] = "gte_cmdw_avsz3",
|
||||
["gte_avg_sort_z4"] = "gte_cmdw_avsz4",
|
||||
["gte_cmdw_sqr"] = "gte_cmdw_sqr",
|
||||
["gte_cmdw_gpf"] = "gte_cmdw_gpf",
|
||||
}
|
||||
|
||||
-- GTE command input-set table.
|
||||
@@ -1136,6 +1138,14 @@ M.GTE_COMMAND_INPUTS = {
|
||||
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||
"gte_cr_ZSF4",
|
||||
},
|
||||
-- SQR: reads IR1..IR3 (per PSX-SPX gte.md SQR section; libgte disassembly 0x800160b0).
|
||||
["gte_cmdw_sqr"] = {
|
||||
"C2_IR1", "C2_IR2", "C2_IR3",
|
||||
},
|
||||
-- GPF: reads IR0 + IR1..IR3 (per PSX-SPX gte.md GPF section; libgte disassembly 0x8001613c).
|
||||
["gte_cmdw_gpf"] = {
|
||||
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||
},
|
||||
}
|
||||
|
||||
-- GTE command output-set + semantic role table.
|
||||
@@ -1208,6 +1218,22 @@ M.GTE_COMMAND_OUTPUTS = {
|
||||
{ register = "C2_IR2", role = "latest_color" },
|
||||
{ register = "C2_IR3", role = "latest_color" },
|
||||
},
|
||||
["gte_cmdw_sqr"] = {
|
||||
{ register = "C2_MAC1", role = "mac_result" },
|
||||
{ register = "C2_MAC2", role = "mac_result" },
|
||||
{ register = "C2_MAC3", role = "mac_result" },
|
||||
{ register = "C2_IR1", role = "latest_color" },
|
||||
{ register = "C2_IR2", role = "latest_color" },
|
||||
{ register = "C2_IR3", role = "latest_color" },
|
||||
},
|
||||
["gte_cmdw_gpf"] = {
|
||||
{ register = "C2_MAC1", role = "mac_result" },
|
||||
{ register = "C2_MAC2", role = "mac_result" },
|
||||
{ register = "C2_MAC3", role = "mac_result" },
|
||||
{ register = "C2_IR1", role = "latest_color" },
|
||||
{ register = "C2_IR2", role = "latest_color" },
|
||||
{ register = "C2_IR3", role = "latest_color" },
|
||||
},
|
||||
}
|
||||
|
||||
-- GTE command/post-command latch-window table.
|
||||
@@ -1270,6 +1296,22 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
|
||||
{ register = "C2_IR2", required = 4 },
|
||||
{ register = "C2_IR3", required = 4 },
|
||||
},
|
||||
["gte_cmdw_sqr"] = {
|
||||
{ register = "C2_MAC1", required = 4 },
|
||||
{ register = "C2_MAC2", required = 4 },
|
||||
{ register = "C2_MAC3", required = 4 },
|
||||
{ register = "C2_IR1", required = 4 },
|
||||
{ register = "C2_IR2", required = 4 },
|
||||
{ register = "C2_IR3", required = 4 },
|
||||
},
|
||||
["gte_cmdw_gpf"] = {
|
||||
{ register = "C2_MAC1", required = 4 },
|
||||
{ register = "C2_MAC2", required = 4 },
|
||||
{ register = "C2_MAC3", required = 4 },
|
||||
{ register = "C2_IR1", required = 4 },
|
||||
{ register = "C2_IR2", required = 4 },
|
||||
{ register = "C2_IR3", required = 4 },
|
||||
},
|
||||
}
|
||||
|
||||
-- Operand-class table for the COP2->GPR load-delay check.
|
||||
@@ -1285,6 +1327,7 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
|
||||
M.OPERAND_READ_POSITIONS = {
|
||||
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
|
||||
["add_ui"] = {1, 2},
|
||||
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
|
||||
["add_ui_self"] = {1},
|
||||
["add_si"] = {1, 2},
|
||||
["add_u"] = {1, 2, 3},
|
||||
@@ -1354,6 +1397,8 @@ M.OPERAND_READ_POSITIONS = {
|
||||
["gte_mv_to_ctrl_r"] = {},
|
||||
["gte_lw"] = {},
|
||||
["gte_sw"] = {},
|
||||
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
|
||||
["shift_aright_var"] = {1, 2, 3},
|
||||
}
|
||||
|
||||
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
|
||||
@@ -1435,8 +1480,10 @@ M.INSTRUCTION_LATENCY = {
|
||||
["xor_i"] = 1, ["xor_u"] = 1,
|
||||
["nor_u"] = 1,
|
||||
["shift_lleft"] = 1, ["shift_lleft_self"] = 1,
|
||||
["shift_lleft_var"] = 1, -- sllv: 1 cycle
|
||||
["shift_lright"] = 1,
|
||||
["shift_aright"] = 1,
|
||||
["shift_aright_var"] = 1, -- srav: 1 cycle
|
||||
["mask_upper"] = 1,
|
||||
["mov_from_high"] = 2, -- mfhi: 2 cycles
|
||||
["mov_from_low"] = 2, -- mflo: 2 cycles
|
||||
@@ -1454,6 +1501,7 @@ M.INSTRUCTION_LATENCY = {
|
||||
["load_half_u"] = 1, ["load_half"] = 1,
|
||||
["load_byte_u"] = 1, ["load_byte"] = 1,
|
||||
["load_upper_i"] = 1,
|
||||
["li_s"] = 1, -- aliased to add_ui(rt, R_0, imm); 1 cycle
|
||||
-- 2-word loads (lui + ori) used for >16-bit immediates
|
||||
["load_imm"] = 2,
|
||||
["load_imm_1w"] = 1,
|
||||
@@ -1497,6 +1545,8 @@ M.INSTRUCTION_LATENCY = {
|
||||
["gte_cmdw_op"] = 6, -- OP: 6 cycles (PSX-SPX)
|
||||
["gte_cmdw_outer_product"] = 6, -- alias for OP
|
||||
["gte_cmdw_wedge"] = 6, -- alias for OP
|
||||
["gte_cmdw_sqr"] = 5, -- SQR(sf): 5 cycles (PSX-SPX); +2 nops for pre-fill if sf=0/1
|
||||
["gte_cmdw_gpf"] = 5, -- GPF(sf,lm): 5 cycles (PSX-SPX); +2 nops for pre-fill if needed
|
||||
-- Long-form aliases (same cycle cost as their short form)
|
||||
["gte_cmdw_rotate_translate_perspective_single"] = 15, -- alias for rtps
|
||||
["gte_cmdw_rotate_translate_perspective_triple"] = 23, -- alias for rtpt
|
||||
@@ -1777,6 +1827,7 @@ M.CU2_TRANSITION_POLICY = {
|
||||
M.INSTRUCTION_GPR_EFFECTS = {
|
||||
-- CPU ALU with one or two GPR operands. Reads every GPR operand position.
|
||||
add_ui = { reads = {1, 2}, writes = {1} },
|
||||
li_s = { reads = {1, 2}, writes = {1} }, -- RMW: rt is both read + written
|
||||
add_ui_self = { reads = {1}, writes = {1} },
|
||||
add_si = { reads = {1, 2}, writes = {1} },
|
||||
add_u = { reads = {2, 3}, writes = {1} },
|
||||
@@ -1893,6 +1944,8 @@ M.INSTRUCTION_GPR_EFFECTS = {
|
||||
atom_writes = { reads = {}, writes = {} },
|
||||
-- mac_yield transfers control to the next atom; zero GPR effects.
|
||||
mac_yield = { reads = {}, writes = {} },
|
||||
shift_lleft_var = { reads = {2, 3}, writes = {1} },
|
||||
shift_aright_var = { reads = {2, 3}, writes = {1} },
|
||||
}
|
||||
|
||||
-- Bounded GPR-value rules consumed by the same forward event walk as `INSTRUCTION_GPR_EFFECTS`.
|
||||
@@ -1903,18 +1956,19 @@ M.INSTRUCTION_GPR_EFFECTS = {
|
||||
-- * passes/static_analysis.lua::apply_gpr_effects
|
||||
-- No second `bounded_value_pass` is permitted.
|
||||
M.GPR_VALUE_RULES = {
|
||||
load_upper_i = { op = "load_upper_i", dest = 1, immediate = 2, },
|
||||
add_ui = { op = "add_ui", dest = 1, source = 2, immediate = 3, },
|
||||
or_i = { op = "or_i", dest = 1, source = 2, immediate = 3, },
|
||||
and_i = { op = "and_i", dest = 1, source = 2, immediate = 3, },
|
||||
xor_i = { op = "xor_i", dest = 1, source = 2, immediate = 3, },
|
||||
add_ui_self = { op = "add_ui", dest = 1, source = 1, immediate = 2, },
|
||||
or_i_self = { op = "or_i", dest = 1, source = 1, immediate = 2, },
|
||||
load_upper_i = { op = "load_upper_i", dest = 1, immediate = 2, },
|
||||
add_ui = { op = "add_ui", dest = 1, source = 2, immediate = 3, },
|
||||
li_s = { op = "add_ui", dest = 1, source = 2, immediate = 3 }, -- R_0 + sign-ext(imm) folds into a constant
|
||||
or_i = { op = "or_i", dest = 1, source = 2, immediate = 3, },
|
||||
and_i = { op = "and_i", dest = 1, source = 2, immediate = 3, },
|
||||
xor_i = { op = "xor_i", dest = 1, source = 2, immediate = 3, },
|
||||
add_ui_self = { op = "add_ui", dest = 1, source = 1, immediate = 2, },
|
||||
or_i_self = { op = "or_i", dest = 1, source = 1, immediate = 2, },
|
||||
-- Present register-form self variants. They are included here so a
|
||||
-- known value is not needlessly lost when these encoders are used.
|
||||
add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, },
|
||||
or_u_self = { op = "or", dest = 1, sources = {1, 2}, },
|
||||
shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, },
|
||||
add_u_self = { op = "add_u", dest = 1, sources = {1, 2}, },
|
||||
or_u_self = { op = "or", dest = 1, sources = {1, 2}, },
|
||||
shift_lleft_self = { op = "shift_lleft", dest = 1, source = 1, immediate = 2, },
|
||||
}
|
||||
|
||||
-- Control-transfer (branch/jump/call) delay-slot policy table.
|
||||
|
||||
@@ -0,0 +1,418 @@
|
||||
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
|
||||
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
|
||||
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
|
||||
-- both parsed ELF32 headers from wire bytes.
|
||||
--
|
||||
-- This module contains the format constants and the byte-level walker.
|
||||
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
|
||||
--
|
||||
-- **Adapter contract (explicit pass style):**
|
||||
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
|
||||
-- so callers wrap once in a 1-line adapter that strips `self`.
|
||||
-- The parsers here operate on the unwrapped form.
|
||||
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
|
||||
-- read_u8(adapter, off) -> integer | nil
|
||||
-- read_u16(adapter, off) -> integer | nil
|
||||
-- read_u32(adapter, off) -> integer | nil
|
||||
-- size(adapter) -> integer
|
||||
--
|
||||
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
|
||||
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
|
||||
--
|
||||
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Little-endian readers (bit-weighted accumulator, math.floor only)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||
---
|
||||
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||
--- byte 0 contributes its value directly;
|
||||
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
|
||||
---
|
||||
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
|
||||
---
|
||||
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
|
||||
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
|
||||
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
|
||||
--- @param adapter table
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer|nil
|
||||
function M.read_u32(adapter, off)
|
||||
return adapter.read_u8_at(off)
|
||||
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||
+ adapter.read_u8_at(off + 0x02) * 0x00010000
|
||||
+ adapter.read_u8_at(off + 0x03) * 0x01000000
|
||||
end
|
||||
|
||||
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||
--- @param adapter table
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer|nil
|
||||
function M.read_u16(adapter, off)
|
||||
return adapter.read_u8_at(off)
|
||||
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||
end
|
||||
|
||||
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||
--- @param adapter table
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer|nil
|
||||
function M.read_u8(adapter, off)
|
||||
return adapter.read_u8_at(off)
|
||||
end
|
||||
|
||||
--- Total adapter byte length.
|
||||
--- @param adapter table
|
||||
--- @return integer
|
||||
function M.size(adapter)
|
||||
return adapter.read_size()
|
||||
end
|
||||
|
||||
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
|
||||
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
|
||||
--- both layers now use the same byte-level helpers under the hood.
|
||||
function M.read_u32_le(buf, off)
|
||||
local byte_off = off + 1
|
||||
return buf:byte(byte_off)
|
||||
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||
end
|
||||
|
||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||
--- @param buf string
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer
|
||||
function M.read_u16_le(buf, off)
|
||||
local byte_off = off + 1
|
||||
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Format constants
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- ELF format constants (System V ABI gABI v1.2).
|
||||
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
|
||||
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
|
||||
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
|
||||
|
||||
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
|
||||
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
|
||||
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
|
||||
|
||||
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
|
||||
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
|
||||
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
|
||||
M.ELF32_HEADER = {
|
||||
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
|
||||
magic = "\127ELF",
|
||||
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
|
||||
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
|
||||
header_bytes = 0x34, -- ELF32 header is 52 bytes total
|
||||
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
|
||||
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||
}
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
|
||||
-- zero-based, field offsets relative to the start of the entry.
|
||||
M.ELF32_SECTION = {
|
||||
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
|
||||
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
|
||||
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
|
||||
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
|
||||
}
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
|
||||
-- zero-based, field offsets relative to the start of the entry.
|
||||
M.ELF32_SYM = {
|
||||
st_name = 0x00, -- 4-byte LE; offset into the linked string table
|
||||
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
|
||||
st_size = 0x08, -- 4-byte LE; symbol size in bytes
|
||||
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
|
||||
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
|
||||
}
|
||||
|
||||
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
|
||||
M.dw_dwarf32_terminator = 0xFFFFFFFF
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Adapter validation
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Validate that `adapter` exposes the byte-read surface.
|
||||
--- Returns true on success, false + a stable error code on failure.
|
||||
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
|
||||
--- @param adapter any
|
||||
--- @return boolean, string|nil
|
||||
function M.validate_adapter(adapter)
|
||||
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
|
||||
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
|
||||
return true, nil
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- String-table reader
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
|
||||
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
|
||||
--- @param strtab string
|
||||
--- @param off integer
|
||||
--- @return string|nil
|
||||
function M.get_str(strtab, off)
|
||||
if off < 0 or off >= #strtab then return nil end
|
||||
local end_pos = strtab:find("\0", off + 1, true)
|
||||
if not end_pos then return nil end
|
||||
return strtab:sub(off + 1, end_pos - 1)
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Header / section / symbol walkers
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
|
||||
--- Returns a table on success:
|
||||
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
|
||||
--- On failure returns nil + a stable error code:
|
||||
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
|
||||
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
|
||||
--- @param adapter table
|
||||
--- @return table|nil, string|nil
|
||||
function M.parse_elf32_headers(adapter)
|
||||
local ok, err = M.validate_adapter(adapter)
|
||||
if not ok then return nil, err end
|
||||
|
||||
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
|
||||
-- The byte readers take the adapter explicitly.
|
||||
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
|
||||
local b1 = M.read_u8(adapter, 0)
|
||||
local b2 = M.read_u8(adapter, 1)
|
||||
local b3 = M.read_u8(adapter, 2)
|
||||
local b4 = M.read_u8(adapter, 3)
|
||||
if not (b1 and b2 and b3 and b4)
|
||||
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
|
||||
return nil, "bad_magic"
|
||||
end
|
||||
|
||||
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset)
|
||||
if class ~= M.ELFCLASS32 then
|
||||
return nil, "unsupported_elf_class"
|
||||
end
|
||||
|
||||
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset)
|
||||
if data ~= M.ELFDATA2LSB then
|
||||
return nil, "unsupported_elf_data"
|
||||
end
|
||||
|
||||
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset)
|
||||
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset)
|
||||
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset)
|
||||
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset)
|
||||
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset)
|
||||
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
|
||||
return nil, "truncated_header"
|
||||
end
|
||||
|
||||
return {
|
||||
e_entry = e_entry,
|
||||
e_shoff = e_shoff,
|
||||
e_shentsize = e_shentsize,
|
||||
e_shnum = e_shnum,
|
||||
e_shstrndx = e_shstrndx,
|
||||
error = nil,
|
||||
}
|
||||
end
|
||||
|
||||
--- Read one section-header entry from `adapter` at `sh_off`.
|
||||
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
|
||||
--- @param adapter table
|
||||
--- @param sh_off integer
|
||||
--- @return table|nil, string|nil -- entry, error
|
||||
local function read_section_entry(adapter, sh_off)
|
||||
local entry = {
|
||||
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
|
||||
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
|
||||
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
|
||||
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
|
||||
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
|
||||
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
|
||||
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
|
||||
name = "",
|
||||
}
|
||||
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
|
||||
and entry.sh_offset and entry.sh_size and entry.sh_link) then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
return entry, nil
|
||||
end
|
||||
|
||||
--- Walk every section header in `hdr` and return a 1-based array of entries
|
||||
--- (the section at logical index 0 is at array position 1, etc.).
|
||||
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
|
||||
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
|
||||
--- @param adapter table
|
||||
--- @param hdr table -- the table returned by parse_elf32_headers
|
||||
--- @return table|nil, string|nil
|
||||
function M.walk_sections(adapter, hdr)
|
||||
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
|
||||
|
||||
local file_size = M.size(adapter)
|
||||
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
|
||||
-- Read every section header first; we need .shstrtab to resolve names.
|
||||
local sections = {}
|
||||
for i = 0, hdr.e_shnum - 1 do
|
||||
local sh_off = hdr.e_shoff + i * hdr.e_shentsize
|
||||
local entry, err = read_section_entry(adapter, sh_off)
|
||||
if not entry then return nil, err end
|
||||
sections[i + 1] = entry
|
||||
end
|
||||
|
||||
if hdr.e_shstrndx >= hdr.e_shnum then
|
||||
return nil, "missing_shstrtab"
|
||||
end
|
||||
|
||||
local shstrtab = sections[hdr.e_shstrndx + 1]
|
||||
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
|
||||
return nil, "missing_shstrtab"
|
||||
end
|
||||
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab)
|
||||
if not shstrtab_bytes then return nil, "truncated_section_headers" end
|
||||
|
||||
for _, s in ipairs(sections) do
|
||||
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
|
||||
end
|
||||
|
||||
return sections, nil
|
||||
end
|
||||
|
||||
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
|
||||
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
|
||||
--- @param adapter table
|
||||
--- @param section table -- one entry from walk_sections
|
||||
--- @return string|nil
|
||||
function M.read_section_bytes(adapter, section)
|
||||
local size = section.sh_size
|
||||
if size == 0 then return "" end
|
||||
local out = {}
|
||||
for i = 0, size - 1 do
|
||||
local b = M.read_u8(adapter, section.sh_offset + i)
|
||||
if b == nil then return nil end
|
||||
out[#out + 1] = string.char(b)
|
||||
end
|
||||
return table.concat(out)
|
||||
end
|
||||
|
||||
--- Convenience: walk sections, then look up the named section, then read its bytes.
|
||||
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
|
||||
--- @param adapter table
|
||||
--- @param sections table -- 1-based array from walk_sections
|
||||
--- @param name string
|
||||
--- @return string|nil, string|nil
|
||||
function M.read_named_section(adapter, sections, name)
|
||||
if not sections then return nil, "missing_section" end
|
||||
for _, s in ipairs(sections) do
|
||||
if s.name == name then
|
||||
local bytes = M.read_section_bytes(adapter, s)
|
||||
if not bytes then return nil, "truncated_section_data" end
|
||||
return bytes, nil
|
||||
end
|
||||
end
|
||||
return nil, "missing_section"
|
||||
end
|
||||
|
||||
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
|
||||
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
|
||||
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
|
||||
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
|
||||
--- @param adapter table
|
||||
--- @param sections table
|
||||
--- @return table|nil, string|nil
|
||||
function M.collect_symbols(adapter, sections)
|
||||
if not sections then return nil, "missing_sections" end
|
||||
local symbols = {}
|
||||
local file_size = M.size(adapter)
|
||||
for _, s in ipairs(sections) do
|
||||
if s.sh_type == M.SHT_SYMTAB then
|
||||
local strtab = sections[s.sh_link + 1]
|
||||
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
|
||||
return nil, "missing_symtab_strtab"
|
||||
end
|
||||
if strtab.sh_offset + strtab.sh_size > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local strtab_bytes = M.read_section_bytes(adapter, strtab)
|
||||
if not strtab_bytes then return nil, "truncated_section_headers" end
|
||||
if s.sh_offset + s.sh_size > file_size then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local symtab_bytes = M.read_section_bytes(adapter, s)
|
||||
if not symtab_bytes then return nil, "truncated_section_headers" end
|
||||
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes
|
||||
for j = 0, n - 1 do
|
||||
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes
|
||||
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name)
|
||||
if st_name then
|
||||
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value)
|
||||
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size)
|
||||
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info)
|
||||
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
|
||||
-- the metaprogram reads too. Inline the read to keep the
|
||||
-- adapter as the only I/O surface.
|
||||
local b1 = M.read_u8(adapter, e + 14)
|
||||
local b2 = M.read_u8(adapter, e + 15)
|
||||
if not (b1 and b2) then
|
||||
return nil, "truncated_section_headers"
|
||||
end
|
||||
local st_shndx = b1 + b2 * 0x100
|
||||
local name = M.get_str(strtab_bytes, st_name) or ""
|
||||
if name ~= "" then
|
||||
symbols[name] = {
|
||||
value = st_value,
|
||||
size = st_size,
|
||||
info = st_info,
|
||||
shndx = st_shndx,
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return symbols, nil
|
||||
end
|
||||
|
||||
return M
|
||||
+153
-140
@@ -11,6 +11,11 @@
|
||||
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
|
||||
local lfs = require("lfs")
|
||||
|
||||
-- scripts/elf32.lua contains format-constant tables + the byte-level walker.
|
||||
-- The this file re-exports `read_u32_le` / `read_u16_le` (and the DWARF32 terminator).
|
||||
-- TODO(Ed): Remove re-export.
|
||||
local E = require("elf32")
|
||||
|
||||
local M = {}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -102,27 +107,13 @@ M.MIPS_BYTES_PER_WORD = 0x04
|
||||
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
|
||||
--- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
|
||||
--- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
|
||||
---
|
||||
--- The ELF32 header / section / sym layout tables are within scripts/elf32.lua.
|
||||
--- The metaprogram re-exports the DWARF32 initial-length terminator.
|
||||
|
||||
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||
M.ELF32 = {
|
||||
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00
|
||||
magic = "\127ELF",
|
||||
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
|
||||
class_elf32 = 1,
|
||||
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
|
||||
endian_little = 1,
|
||||
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
|
||||
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
|
||||
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
||||
}
|
||||
--- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
||||
M.dw_dwarf32_terminator = E.dw_dwarf32_terminator
|
||||
-- TODO(Ed): Remove re-export.
|
||||
|
||||
-- ----------------------------------------------------------------------------
|
||||
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
|
||||
@@ -241,27 +232,24 @@ M.DWARF5_DEBUG_LINE = {
|
||||
--- (which has partial `string.unpack` coverage).
|
||||
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
|
||||
---
|
||||
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||
--- byte 0 contributes its value directly; byte 1 is shifted left by 8 (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000).
|
||||
--- Thin forwarder: the canonical implementation lives in scripts/elf32.lua.
|
||||
--- The "second caller lifts" pattern keeps the metaprogram side fluent
|
||||
--- (`M.read_u32_le(buf, off)`) while the body is deduped.
|
||||
--- @param buf string
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer
|
||||
function M.read_u32_le(buf, off)
|
||||
local byte_off = off + 1
|
||||
return buf:byte(byte_off)
|
||||
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||
return E.read_u32_le(buf, off)
|
||||
end
|
||||
|
||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
|
||||
--- Thin forwarder — see `M.read_u32_le` for the rationale.
|
||||
--- @param buf string
|
||||
--- @param off integer -- zero-based wire offset
|
||||
--- @return integer
|
||||
function M.read_u16_le(buf, off)
|
||||
local byte_off = off + 1
|
||||
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||
return E.read_u16_le(buf, off)
|
||||
end
|
||||
|
||||
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
|
||||
@@ -442,20 +430,20 @@ function M.read_ref_sig8(buf, pos)
|
||||
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
|
||||
end
|
||||
|
||||
-- DWARF5 §7.5.6 (Type Entries).
|
||||
-- Walk all units in `info` and return the 0-based offset of the first unit
|
||||
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`.
|
||||
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
||||
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
||||
--
|
||||
-- Unit header layout (from pos 0):
|
||||
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
||||
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
||||
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
||||
-- @param info string -- the .debug_info section bytes
|
||||
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
||||
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
||||
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
||||
--- DWARF5 §7.5.6 (Type Entries).
|
||||
--- Walk all units in `info` and return the 0-based offset of the first unit whose `DW_AT_type_signature`
|
||||
--- (8-byte value at the end of the unit header) equals `target_sig`.
|
||||
--- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
||||
--- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
||||
---
|
||||
--- Unit header layout (from pos 0):
|
||||
--- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
||||
--- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
||||
--- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
||||
--- @param info string -- the .debug_info section bytes
|
||||
--- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
||||
--- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
||||
--- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
||||
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
||||
local pos = 0
|
||||
local section_len = #info
|
||||
@@ -539,7 +527,7 @@ end
|
||||
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
|
||||
--- For frequent callers, pass the union of all needed sections in one call.
|
||||
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
|
||||
--- @param elf_path Path
|
||||
--- @param elf_path Path
|
||||
--- @param section_names string[] -- list of section names to read
|
||||
--- @return table<string, string>
|
||||
function M.read_elf_sections(elf_path, section_names)
|
||||
@@ -564,69 +552,58 @@ function M.read_elf_sections(elf_path, section_names)
|
||||
return result
|
||||
end
|
||||
|
||||
-- Read the ELF32 header.
|
||||
local header = f:read(M.ELF32.header_bytes)
|
||||
if not header or #header < M.ELF32.header_bytes then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n")
|
||||
local file_size
|
||||
do
|
||||
f:seek("end", 0)
|
||||
file_size = f:seek("cur", 0)
|
||||
end
|
||||
local adapter = {
|
||||
read_u8_at = function(offset)
|
||||
f:seek("set", offset)
|
||||
local b = f:read(1)
|
||||
if not b then return nil end
|
||||
return b:byte()
|
||||
end,
|
||||
read_u16_at = function(offset)
|
||||
f:seek("set", offset)
|
||||
local b1 = f:read(1)
|
||||
local b2 = f:read(1)
|
||||
if not b1 or not b2 then return nil end
|
||||
return b1:byte() + b2:byte() * 0x100
|
||||
end,
|
||||
read_u32_at = function(offset)
|
||||
f:seek("set", offset)
|
||||
local b1 = f:read(1)
|
||||
local b2 = f:read(1)
|
||||
local b3 = f:read(1)
|
||||
local b4 = f:read(1)
|
||||
if not b1 or not b2 or not b3 or not b4 then return nil end
|
||||
return b1:byte() + b2:byte() * 0x100
|
||||
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
|
||||
end,
|
||||
read_size = function() return file_size end,
|
||||
}
|
||||
|
||||
-- Delegate the header parse + section walk to E.*.
|
||||
local hdr, hdr_err = E.parse_elf32_headers(adapter)
|
||||
if not hdr then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] header parse failed: %s\n", tostring(hdr_err)))
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
|
||||
-- Sanity-check magic + class + endianness.
|
||||
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n")
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
|
||||
local sections, walk_err = E.walk_sections(adapter, hdr)
|
||||
if not sections then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] section walk failed: %s\n", tostring(walk_err)))
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
|
||||
-- Parse section-header table location + dimensions from the header.
|
||||
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset)
|
||||
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset)
|
||||
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset)
|
||||
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset)
|
||||
|
||||
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
|
||||
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
|
||||
local strtab_hdr = f:read(e_shentsize)
|
||||
if not strtab_hdr or #strtab_hdr < e_shentsize then
|
||||
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
|
||||
f:close()
|
||||
return result
|
||||
end
|
||||
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
|
||||
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
|
||||
f:seek("set", strtab_offset)
|
||||
local strtab = f:read(strtab_size) or ""
|
||||
|
||||
-- Walk all section headers; collect (offset, size) for the wanted names.
|
||||
local function read_section_bytes(sh_offset, sh_size)
|
||||
f:seek("set", sh_offset)
|
||||
return f:read(sh_size) or ""
|
||||
end
|
||||
|
||||
for sh_idx = 0, e_shnum - 1 do
|
||||
f:seek("set", e_shoff + sh_idx * e_shentsize)
|
||||
local sh = f:read(e_shentsize)
|
||||
if not sh or #sh < e_shentsize then break end
|
||||
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
|
||||
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
|
||||
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
|
||||
|
||||
-- Extract the name (null-terminated C string in strtab).
|
||||
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
|
||||
local name = strtab:sub(sh_name + 1, name_end - 1)
|
||||
if wanted[name] then
|
||||
result[name] = read_section_bytes(sh_offset, sh_size)
|
||||
-- Resolve the requested sections.
|
||||
for _, s in ipairs(sections) do
|
||||
if wanted[s.name] then
|
||||
local bytes = E.read_section_bytes(adapter, s)
|
||||
if bytes then result[s.name] = bytes end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -643,48 +620,87 @@ end
|
||||
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
||||
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
||||
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
||||
---
|
||||
--- @param elf_path Path
|
||||
--- @return table<string, {integer, integer}>
|
||||
function M.read_nm(elf_path)
|
||||
local addrs = {}
|
||||
|
||||
-- Read .symtab + .strtab via the existing ELF walker (no subprocess).
|
||||
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"})
|
||||
local symtab = sections[".symtab"]
|
||||
local strtab = sections[".strtab"]
|
||||
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
|
||||
-- No symbol table (e.g. stripped ELF). Return empty.
|
||||
-- Existence check first; an empty or missing ELF returns an empty map.
|
||||
if lfs.attributes(elf_path, "mode") ~= "file" then
|
||||
return addrs
|
||||
end
|
||||
|
||||
-- Iterate the 16-byte ELF32 symtab entries.
|
||||
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14.
|
||||
local SYM_ENTRY_BYTES = 0x10
|
||||
local SYM_ST_NAME = 0x00
|
||||
local SYM_ST_VALUE = 0x04
|
||||
local SYM_ST_SIZE = 0x08
|
||||
local SYM_ST_INFO = 0x0C
|
||||
local n_syms = #symtab / SYM_ENTRY_BYTES
|
||||
for i = 0, n_syms - 1 do
|
||||
local entry_off = i * SYM_ENTRY_BYTES
|
||||
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1)
|
||||
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
||||
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions).
|
||||
local binding = math.floor(st_info / 16)
|
||||
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL
|
||||
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE)
|
||||
if st_size > 0 then
|
||||
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME)
|
||||
-- Extract the name from .strtab (null-terminated C string).
|
||||
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
|
||||
local name = strtab:sub(st_name_off + 1, name_end - 1)
|
||||
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
|
||||
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
|
||||
if name and #name > 0 then
|
||||
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
|
||||
addrs[name] = { st_value, st_size }
|
||||
end
|
||||
end
|
||||
local f = io.open(elf_path, "rb")
|
||||
if not f then
|
||||
return addrs
|
||||
end
|
||||
|
||||
-- Build the file adapter for E.*.
|
||||
local file_size
|
||||
do
|
||||
f:seek("end", 0)
|
||||
file_size = f:seek("cur", 0)
|
||||
end
|
||||
local adapter = {
|
||||
read_u8_at = function(offset)
|
||||
f:seek("set", offset)
|
||||
local b = f:read(1)
|
||||
if not b then return nil end
|
||||
return b:byte()
|
||||
end,
|
||||
read_u16_at = function(offset)
|
||||
f:seek("set", offset)
|
||||
local b1 = f:read(1)
|
||||
local b2 = f:read(1)
|
||||
if not b1 or not b2 then return nil end
|
||||
return b1:byte() + b2:byte() * 0x100
|
||||
end,
|
||||
read_u32_at = function(offset)
|
||||
f:seek("set", offset)
|
||||
local b1 = f:read(1)
|
||||
local b2 = f:read(1)
|
||||
local b3 = f:read(1)
|
||||
local b4 = f:read(1)
|
||||
if not b1 or not b2 or not b3 or not b4 then return nil end
|
||||
return b1:byte() + b2:byte() * 0x100
|
||||
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
|
||||
end,
|
||||
read_size = function() return file_size end,
|
||||
}
|
||||
|
||||
-- Delegate the header + section walk to E.*.
|
||||
local hdr, hdr_err = E.parse_elf32_headers(adapter)
|
||||
if not hdr then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_nm] header parse failed: %s\n", tostring(hdr_err)))
|
||||
f:close()
|
||||
return addrs
|
||||
end
|
||||
|
||||
local sections, walk_err = E.walk_sections(adapter, hdr)
|
||||
if not sections then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_nm] section walk failed: %s\n", tostring(walk_err)))
|
||||
f:close()
|
||||
return addrs
|
||||
end
|
||||
|
||||
-- E.collect_symbols returns every defined symbol (no binding filter).
|
||||
-- The metaprogram then applies its STB_LOCAL / STB_GLOBAL + size>0 filter, matching `nm`'s default (external symbols only).
|
||||
local symbols, sym_err = E.collect_symbols(adapter, sections)
|
||||
if not symbols then
|
||||
io.stderr:write(string.format("[elf_dwarf.read_nm] symbol collection failed: %s\n", tostring(sym_err)))
|
||||
f:close()
|
||||
return addrs
|
||||
end
|
||||
|
||||
f:close()
|
||||
|
||||
for name, entry in pairs(symbols) do
|
||||
-- High nibble of st_info = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
||||
-- math.floor(/16) is portable across LuaJIT 2.0/2.1 and plain Lua 5.x.
|
||||
local binding = math.floor(entry.info / 16)
|
||||
if (binding == 0 or binding == 1) and entry.size > 0 then
|
||||
addrs[name] = { entry.value, entry.size }
|
||||
end
|
||||
end
|
||||
|
||||
@@ -822,12 +838,11 @@ end
|
||||
--- * The `.debug_line` section may contain MULTIPLE line-program units
|
||||
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
|
||||
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
|
||||
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom
|
||||
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
|
||||
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom `inv.call_file`
|
||||
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
|
||||
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
|
||||
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index`
|
||||
--- emits 2 forms: path + dir_index). The helper supports:
|
||||
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` emits 2 forms: path + dir_index).
|
||||
--- The helper supports:
|
||||
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
|
||||
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
|
||||
--- - DW_FORM_udata (ULEB128)
|
||||
@@ -837,9 +852,7 @@ end
|
||||
---
|
||||
--- Behavior on failure: writes to stderr and returns nil.
|
||||
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
|
||||
--- downstream `resolve_provenance_file_index(path)` consumers
|
||||
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
|
||||
--- consult the map directly.
|
||||
--- downstream `resolve_provenance_file_index(path)` consumers consult the map directly.
|
||||
---
|
||||
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
|
||||
--- @return table|nil, table|nil, table|nil
|
||||
|
||||
@@ -0,0 +1,322 @@
|
||||
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
|
||||
---
|
||||
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
|
||||
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
|
||||
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
|
||||
---
|
||||
--- User-pinned GPRs (added 2026-08-10): The corpus's `register_alias_registry` is consulted to
|
||||
--- exclude GPRs the user has pinned via `atom_reg` + `_Code` defs (e.g. wave-context carriers like
|
||||
--- `R_ResolveScratch = R_T4 atom_reg`). These GPRs are unavailable to EVERY atom's source pool,
|
||||
--- not just to atoms in the same phase — wave-context carriers are preserved across atoms by the
|
||||
--- wave-context discipline and must never be reallocated.
|
||||
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
|
||||
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will
|
||||
--- exclude R_T4 from that atom's pool.
|
||||
---
|
||||
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
|
||||
--- emit `phase_register_clash` as an info finding (no build stop). Should be unreachable after the
|
||||
--- user-pinning + body-parsing fix above; kept as a defensive safety net.
|
||||
---
|
||||
--- Pool exhaustion: if a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
|
||||
--- emit `phase_register_pool_exhausted` as a build-stopping error.
|
||||
---
|
||||
--- @class AutoRegResult
|
||||
--- @field outputs table[] -- {kind=, path=} entries
|
||||
--- @field errors table[] -- {line=, msg=} entries (build-stops)
|
||||
--- @field warnings table[] -- {line=, msg=} entries (build-continues)
|
||||
|
||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||
|
||||
-- The fixed allocation pool: 10 physical GPRs whose `R_<Sym>_Code` macros exist in mips.h (lines 92-107).
|
||||
-- Each pool entry is the PHYSICAL GPR ident (R_T0 etc.);
|
||||
-- `gpr .. "_Code"` resolves to the matching `R_Tn_Code` constant the source code references via `#define R_Load_Code R_T0_Code`.
|
||||
-- Excluded: R_AT (assembler temp per lottes_tape.h:86), R_T8 (deferred to ac_yield_load pattern),
|
||||
-- R_T9 (R_TapePtr; owned by the tape runtime).
|
||||
local POOL = {
|
||||
"R_T0", "R_T1", "R_T2", "R_T3",
|
||||
"R_T4", "R_T5", "R_T6", "R_T7",
|
||||
"R_V0", "R_V1",
|
||||
}
|
||||
|
||||
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident
|
||||
-- in POOL. The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines
|
||||
-- (mips.h:92-123). Only the POOL entries matter for auto_reg — non-pool aliases are out of scope.
|
||||
local INT_CODE_TO_POOL_GPR = {
|
||||
[2] = "R_V0", [3] = "R_V1",
|
||||
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
||||
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
||||
}
|
||||
|
||||
-- Stable sort for deterministic allocation order.
|
||||
local function stable_sort_keys(tbl)
|
||||
local keys = {}
|
||||
for k in pairs(tbl) do keys[#keys + 1] = k end
|
||||
table.sort(keys)
|
||||
return keys
|
||||
end
|
||||
|
||||
-- Allocate one phase's auto-reg mappings.
|
||||
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
|
||||
local function allocate_phase(phase_label, decls)
|
||||
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
|
||||
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
|
||||
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
|
||||
local pool = {}
|
||||
for i = 1, #POOL do pool[i] = POOL[i] end
|
||||
local result = {}
|
||||
local errors = {}
|
||||
for _, sym in ipairs(stable_sort_keys(decls)) do
|
||||
local next_gpr = table.remove(pool, 1)
|
||||
if not next_gpr then
|
||||
errors[#errors + 1] = {
|
||||
line = 0,
|
||||
msg = string.format(
|
||||
"phase_register_pool_exhausted: phase '%s' requested symbol '%s' but the pool has no remaining registers (max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
|
||||
, phase_label, sym),
|
||||
}
|
||||
return result, errors
|
||||
end
|
||||
result[sym] = next_gpr
|
||||
end
|
||||
return result, errors
|
||||
end
|
||||
|
||||
-- Build two projections from corpus.register_alias_registry:
|
||||
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally
|
||||
-- -- (wave-context carriers, file-scope pinned aliases)
|
||||
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
|
||||
-- Both projections are derived from the same set of entries: every AliasEntry in
|
||||
-- register_alias_registry has `has_atom_reg = true` (only those entries are added to the
|
||||
-- registry; see passes/scan_source.lua parse_enum_entry). Each entry's `code` is the integer
|
||||
-- MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
|
||||
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored — they
|
||||
-- don't affect the auto_reg pool, and they're already excluded from POOL above.
|
||||
local function build_user_pins(corpus)
|
||||
local user_pinned = {}
|
||||
local alias_to_gpr = {}
|
||||
if not corpus.register_alias_registry then
|
||||
return user_pinned, alias_to_gpr
|
||||
end
|
||||
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do
|
||||
if alias_entry.has_atom_reg and alias_entry.code then
|
||||
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code]
|
||||
if gpr then
|
||||
user_pinned[gpr] = true
|
||||
alias_to_gpr[alias_name] = gpr
|
||||
end
|
||||
end
|
||||
end
|
||||
return user_pinned, alias_to_gpr
|
||||
end
|
||||
|
||||
-- Find every physical GPR referenced in the atom body, via EITHER:
|
||||
-- (a) a hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
|
||||
-- (b) an alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
|
||||
-- Returns { [physical_gpr_ident] = count }. The clash-detection and source-pool-exclusion logic
|
||||
-- only needs the presence of each GPR (boolean test), but keeping the count preserves the
|
||||
-- original find_hardcoded_rn shape so callers can switch without churn.
|
||||
-- The alias pattern is sorted lexicographically to keep the regex deterministic.
|
||||
local function find_used_gprs(body_text, alias_to_gpr)
|
||||
local found = {}
|
||||
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
|
||||
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do
|
||||
found[gpr] = (found[gpr] or 0) + 1
|
||||
end
|
||||
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
|
||||
-- Sorted by name so the regex is byte-stable across runs.
|
||||
if alias_to_gpr and next(alias_to_gpr) then
|
||||
local aliases = {}
|
||||
for alias_name in pairs(alias_to_gpr) do
|
||||
aliases[#aliases + 1] = alias_name
|
||||
end
|
||||
table.sort(aliases)
|
||||
local pattern = "(" .. table.concat(aliases, "|") .. ")"
|
||||
for alias_name in body_text:gmatch(pattern) do
|
||||
local gpr = alias_to_gpr[alias_name]
|
||||
if gpr and not found[gpr] then
|
||||
found[gpr] = 1
|
||||
end
|
||||
end
|
||||
end
|
||||
return found
|
||||
end
|
||||
|
||||
-- Emit one gen/auto_reg.h header per directory.
|
||||
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
|
||||
if not mappings or next(mappings) == nil then return end
|
||||
local out_path = out_dir .. "/" .. "auto_reg.h"
|
||||
duffle.ensure_dir(out_dir)
|
||||
local lines = {
|
||||
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||
"#pragma once",
|
||||
"#endif",
|
||||
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
|
||||
"// Directory: " .. dir:gsub("/", "\\"),
|
||||
}
|
||||
for _, src in ipairs(sources) do
|
||||
lines[#lines + 1] = "// source: " .. src.path
|
||||
end
|
||||
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
|
||||
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
|
||||
lines[#lines + 1] = ""
|
||||
for _, sym in ipairs(stable_sort_keys(mappings)) do
|
||||
local gpr = mappings[sym]
|
||||
local gpr_code = gpr .. "_Code"
|
||||
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
|
||||
end
|
||||
lines[#lines + 1] = ""
|
||||
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||
print(" -> " .. out_path)
|
||||
return out_path
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- Pass entry
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
local M = {}
|
||||
|
||||
--- @param ctx PassCtx
|
||||
--- @return AutoRegResult
|
||||
function M.run(ctx)
|
||||
local outputs = {}
|
||||
local errors = {}
|
||||
local warnings = {}
|
||||
|
||||
local corpus = ctx.shared and ctx.shared.corpus
|
||||
if type(corpus) ~= "table" then
|
||||
error("auto_reg.run requires ctx.shared.corpus", 0)
|
||||
end
|
||||
|
||||
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
|
||||
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in
|
||||
-- hello_camera.atom.c) MUST NOT be allocated to any auto-reg marker — they're
|
||||
-- preserved across atoms by the wave-context discipline. The corpus's
|
||||
-- register_alias_registry is the source of truth for these opt-in pins.
|
||||
-- Body references to those aliases (via alias_to_gpr) are also excluded on a
|
||||
-- per-atom basis in step 2 below.
|
||||
local user_pinned, alias_to_gpr = build_user_pins(corpus)
|
||||
|
||||
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
|
||||
local phase_allocations = {}
|
||||
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do
|
||||
local mapping, errs = allocate_phase(phase_label, decls)
|
||||
for sym, gpr in pairs(mapping) do
|
||||
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
|
||||
phase_allocations[phase_label][sym] = gpr
|
||||
end
|
||||
for _, e in ipairs(errs) do
|
||||
errors[#errors + 1] = e
|
||||
end
|
||||
end
|
||||
|
||||
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
|
||||
-- Otherwise, allocate a private pool for the atom.
|
||||
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)` in the atom's `atom_info` line).
|
||||
-- Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
|
||||
local atom_name_to_phase = {}
|
||||
for phase_label, entry in pairs(corpus.atom_phases or {}) do
|
||||
for _, atom_name in ipairs(entry.atoms or {}) do
|
||||
atom_name_to_phase[atom_name] = phase_label
|
||||
end
|
||||
end
|
||||
|
||||
local atom_allocations = {}
|
||||
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do
|
||||
local phase_label = atom_name_to_phase[atom_scope]
|
||||
-- Build the atom's source pool: start with the full POOL, subtract:
|
||||
-- (a) every GPR already committed (phase allocations + prior atom allocations)
|
||||
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
|
||||
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
|
||||
-- (the latter resolved via alias_to_gpr; this catches cases where the user
|
||||
-- wrote R_ResolveScratch instead of R_T4 directly)
|
||||
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
|
||||
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
|
||||
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
|
||||
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
|
||||
local used = {}
|
||||
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
|
||||
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
|
||||
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
|
||||
-- Folded into `used` so the source_pool exclusion is a single check.
|
||||
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
|
||||
if atom and atom.body then
|
||||
local body_used = find_used_gprs(atom.body, alias_to_gpr)
|
||||
for gpr in pairs(body_used) do used[gpr] = true end
|
||||
end
|
||||
local source_pool = {}
|
||||
for _, gpr in ipairs(POOL) do
|
||||
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers
|
||||
-- declared via atom_reg + _Code defs, preserved across atoms globally).
|
||||
if not used[gpr] and not user_pinned[gpr] then
|
||||
source_pool[#source_pool + 1] = gpr
|
||||
end
|
||||
end
|
||||
local result = {}
|
||||
for _, sym in ipairs(stable_sort_keys(decls)) do
|
||||
local next_gpr = table.remove(source_pool, 1)
|
||||
if not next_gpr then
|
||||
errors[#errors + 1] = {
|
||||
line = 0,
|
||||
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' but no free registers remain in its scope pool."
|
||||
, atom_scope, sym),
|
||||
}
|
||||
else
|
||||
result[sym] = next_gpr
|
||||
end
|
||||
end
|
||||
atom_allocations[atom_scope] = result
|
||||
end
|
||||
|
||||
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
|
||||
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
|
||||
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>). An auto-reg allocation that
|
||||
-- matched an existing body reference would be impossible by construction. This warning is kept
|
||||
-- as a defensive safety net for cases the body scanner might miss (e.g. macros that expand to
|
||||
-- register references the scanner cannot resolve).
|
||||
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
|
||||
for atom_scope, decls in pairs(atom_allocations) do
|
||||
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
|
||||
if atom and atom.body then
|
||||
local used_in_body = find_used_gprs(atom.body, alias_to_gpr)
|
||||
for sym, allocated_gpr in pairs(decls) do
|
||||
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
|
||||
warnings[#warnings + 1] = {
|
||||
line = atom.line or 0,
|
||||
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
|
||||
, atom_scope, allocated_gpr, sym, allocated_gpr),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- 4. Emit per-directory gen/auto_reg.h.
|
||||
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
|
||||
local sources_by_dir = corpus.sources_by_dir or {}
|
||||
for dir, sources in pairs(sources_by_dir) do
|
||||
local per_dir_mappings = {}
|
||||
for _, src in ipairs(sources) do
|
||||
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
|
||||
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
|
||||
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable, which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
|
||||
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do
|
||||
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do
|
||||
per_dir_mappings[sym] = gpr
|
||||
end
|
||||
end
|
||||
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do
|
||||
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do
|
||||
per_dir_mappings[sym] = gpr
|
||||
end
|
||||
end
|
||||
end
|
||||
local out_dir = dir .. "/gen"
|
||||
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings)
|
||||
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
|
||||
end
|
||||
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||
end
|
||||
|
||||
return M
|
||||
@@ -3,7 +3,7 @@
|
||||
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
|
||||
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
||||
---
|
||||
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
|
||||
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)`, `MipsAtomComp_Proc_(ac_X, { body })`, and `MipsAtom_Proc_(X, ab, { body })` declarations,
|
||||
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||
---
|
||||
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||
@@ -76,7 +76,7 @@ local MACS_FILENAME = "macs.h"
|
||||
--- @field args string|nil -- Function-args string (function form only)
|
||||
--- @field line integer -- Source line of the declaration
|
||||
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||
--- @field kind string -- "comp_bare" | "comp_proc" | "atom_proc"
|
||||
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -200,8 +200,16 @@ end
|
||||
local function project_components(source, scan)
|
||||
local out = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
||||
if a.kind == "comp_bare" or a.kind == "comp_proc" or a.kind == "atom_proc" then
|
||||
-- `MipsAtom_Proc_` atoms have no `FI_ Slice_MipsCode ac_X(...)` function-decl prelude
|
||||
-- (the macro sits inside a wrapping `I_ void <proc_name>(...)` body), so the function-args
|
||||
-- lookup is meaningless; signature defaults to `...` (variadic-ignored).
|
||||
-- The `mac_<name>` alias expansion discards the `ab` (atom-builder) arg the same way
|
||||
-- `MipsAtomComp_Proc_` components do.
|
||||
local args = nil
|
||||
if a.kind ~= "atom_proc" then
|
||||
args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
||||
end
|
||||
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
||||
-- The pass reads `declaration_comment` directly.
|
||||
local comment = a.declaration_comment or ""
|
||||
@@ -213,7 +221,7 @@ local function project_components(source, scan)
|
||||
body_tokens = a.body_tokens,
|
||||
args = args,
|
||||
comment = comment,
|
||||
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
||||
kind = a.kind, -- "comp_bare" | "comp_proc" | "atom_proc"; provenance emitter reads this.
|
||||
debug_skip = a.debug_skip == true,
|
||||
}
|
||||
end
|
||||
@@ -299,7 +307,9 @@ local function word_count_rec(name, comp_by_name, wc, cache)
|
||||
local trimmed = t.tok
|
||||
if trimmed ~= "" then
|
||||
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
||||
if lookup and comp_by_name[lookup] then
|
||||
if lookup == "atom_label" or lookup == "atom_offset" then
|
||||
-- Pure metaprogram anchors; emit zero words.
|
||||
elseif lookup and comp_by_name[lookup] then
|
||||
-- It's a `mac_X(...)` call. Recurse.
|
||||
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
||||
elseif lookup and wc and wc[lookup] then
|
||||
@@ -473,12 +483,26 @@ local function split_comment_lines(s)
|
||||
end
|
||||
|
||||
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
||||
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
|
||||
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies
|
||||
--- don't reference `ab` (the builder is only consumed by the procedural `atombuilder_unroll` line
|
||||
--- that `MipsAtomComp_Proc_` appends after the body). Inline callers therefore don't need to thread
|
||||
--- a builder context.
|
||||
--- @param args_str string|nil
|
||||
--- @return string
|
||||
local function signature_from_args(args_str)
|
||||
local arg_names = extract_arg_names(args_str)
|
||||
if arg_names and #arg_names > 0 then
|
||||
return table.concat(arg_names, ", ")
|
||||
-- Drop the leading `ab` (atom-builder) first arg if present.
|
||||
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
|
||||
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
|
||||
if arg_names[1] == "ab" then
|
||||
table.remove(arg_names, 1)
|
||||
end
|
||||
if #arg_names > 0 then
|
||||
return table.concat(arg_names, ", ")
|
||||
end
|
||||
return "..." -- `ab` was the only arg; fall through to variadic
|
||||
end
|
||||
return "..."
|
||||
end
|
||||
@@ -644,7 +668,7 @@ end
|
||||
--- @field name string -- bare name (without ac_/mac_ prefix)
|
||||
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
|
||||
--- @field path string -- absolute source path of the definition
|
||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||
--- @field kind string -- "comp_bare" | "comp_proc" | "atom_proc"
|
||||
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
|
||||
|
||||
--- (internal) Populate `corpus.components` with this source's components-by-name map.
|
||||
|
||||
@@ -992,7 +992,7 @@ local function build_dwarf_line_section(existing, atom_table)
|
||||
while unit_pos < #existing do
|
||||
if unit_pos + 4 > #existing then return existing end
|
||||
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
|
||||
if unit_length == elf_dwarf.ELF32.dw_dwarf32_terminator then return existing end
|
||||
if unit_length == elf_dwarf.dw_dwarf32_terminator then return existing end
|
||||
local unit_end_excl = unit_pos + 4 + unit_length
|
||||
if unit_end_excl > #existing then return existing end
|
||||
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
|
||||
@@ -1053,7 +1053,7 @@ local function build_dwarf_aranges_section(existing, atom_table)
|
||||
while i < #existing do
|
||||
-- Read this unit's length.
|
||||
local ul = elf_dwarf.read_u32_le(existing, i)
|
||||
if ul == elf_dwarf.ELF32.dw_dwarf32_terminator then
|
||||
if ul == elf_dwarf.dw_dwarf32_terminator then
|
||||
-- DWARF64 marker - not supported.
|
||||
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
|
||||
return existing
|
||||
|
||||
@@ -188,11 +188,11 @@ function M.run(ctx)
|
||||
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
||||
|
||||
-- Project once, collect errors + warnings for one atom.
|
||||
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
|
||||
-- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
|
||||
local function process_atom(atom, src)
|
||||
if not (atom and atom.body) then return end
|
||||
local kind = atom.kind
|
||||
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||
if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||
return
|
||||
end
|
||||
local proj = project_atom(atom, src, corpus)
|
||||
@@ -215,7 +215,7 @@ function M.run(ctx)
|
||||
end
|
||||
|
||||
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
||||
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||
-- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
||||
for _, src in ipairs(corpus.source_order) do
|
||||
local scan = src.scan or {}
|
||||
|
||||
+210
-13
@@ -3,6 +3,7 @@
|
||||
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once,
|
||||
--- extracting every construct type the metaprograms need:
|
||||
--- MipsAtom_ (kind = "atom", with optional atom_info inner)
|
||||
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
|
||||
--- MipsAtomComp_ (kind = "comp_bare")
|
||||
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
|
||||
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
|
||||
@@ -34,7 +35,7 @@ local parse_enum_int_literal
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
--- @class SourceScan
|
||||
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtomComp_ + MipsAtomComp_Proc_
|
||||
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtom_Proc_ + MipsAtomComp_ + MipsAtomComp_Proc_
|
||||
--- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only)
|
||||
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
|
||||
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
|
||||
@@ -55,7 +56,7 @@ local parse_enum_int_literal
|
||||
--- @field args string|nil -- Trimmed args inside the `(...)` (nil when has_parens is false)
|
||||
--- @field pending boolean -- true while awaiting the following declaration
|
||||
--- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot
|
||||
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
|
||||
--- @field target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
|
||||
--- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_`
|
||||
|
||||
--- @class RegTypeDefault
|
||||
@@ -111,7 +112,7 @@ local parse_enum_int_literal
|
||||
--- @field name string -- Atom name (for components: without ac_ prefix)
|
||||
--- @field body string -- Brace-delimited body (without the braces)
|
||||
--- @field body_off integer -- Char offset of body[1] in source
|
||||
--- @field kind string -- "atom" | "comp_bare" | "comp_proc" | "raw_atom"
|
||||
--- @field kind string -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "raw_atom"
|
||||
--- @field raw_name string -- Un-stripped name (for components: with ac_ prefix)
|
||||
--- @field ident_pos integer -- Position of the MipsAtom_/MipsAtomComp_ ident start
|
||||
--- @field after_paren integer -- Position past the closing paren
|
||||
@@ -268,7 +269,7 @@ end
|
||||
--- marker_kind == "atom_dbg_skip" AND is_bare == true
|
||||
--- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`.
|
||||
--- @param out SourceScan
|
||||
--- @param target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
||||
--- @param target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
||||
--- @return boolean|nil -- true iff the marker is the positive bare form
|
||||
local function attach_debug_skip_marker(out, target_kind)
|
||||
local markers = out.debug_skip_markers
|
||||
@@ -799,6 +800,11 @@ local BYTE_x = 0x78 -- 'x'
|
||||
local BYTE_X = 0x58 -- 'X'
|
||||
local BYTE_OPEN_BRACE = 0x7B -- '{'
|
||||
local BYTE_CLOSE_BRACE= 0x7D -- '}'
|
||||
local BYTE_SLASH = 0x2F -- '/'
|
||||
local BYTE_STAR = 0x2A -- '*'
|
||||
local BYTE_SPACE = 0x20 -- ' '
|
||||
local BYTE_TAB = 0x09 -- '\t'
|
||||
local BYTE_CR = 0x0D -- '\r'
|
||||
|
||||
-- Maximum chain depth when resolving `R_*_Code` symbol RHS references.
|
||||
-- Eight hops is enough for any production chain (R_TapePtr_Code -> R_T8_Code -> ...).
|
||||
@@ -822,6 +828,44 @@ local function hex_digit_value(b)
|
||||
return nil
|
||||
end
|
||||
|
||||
-- Read one trailing C-comment that appears immediately after `pos` in `body`,
|
||||
-- skipping horizontal whitespace and newlines first. Used by `parse_enum_entry` to
|
||||
-- recover the `atom_auto_reg:` / `phase_auto_reg:` scope annotation embedded by
|
||||
-- the `atom_auto_reg` / `phase_auto_reg` macros' RHS expansion
|
||||
-- (`R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`).
|
||||
-- Handles both block (`/* ... */`) and line (`// ...`) forms.
|
||||
-- Returns the comment text (without delimiters), or nil if no comment is adjacent.
|
||||
local function read_trailing_cmt_after(body, pos)
|
||||
local body_len = #body
|
||||
while pos <= body_len do
|
||||
local b = body:byte(pos)
|
||||
if b == BYTE_SPACE or b == BYTE_TAB or b == BYTE_NEWLINE or b == BYTE_CR then
|
||||
pos = pos + 1
|
||||
elseif b == BYTE_SLASH then
|
||||
local b2 = body:byte(pos + 1)
|
||||
if b2 == BYTE_STAR then
|
||||
-- Block comment /* ... */
|
||||
local i = pos + 2
|
||||
while i < body_len do
|
||||
if body:byte(i) == BYTE_STAR and body:byte(i + 1) == BYTE_SLASH then
|
||||
return body:sub(pos + 2, i - 1)
|
||||
end
|
||||
i = i + 1
|
||||
end
|
||||
return nil -- unterminated; treat as no comment
|
||||
elseif b2 == BYTE_SLASH then
|
||||
-- Line comment // ... (strip the trailing newline)
|
||||
local end_pos = duffle.find_byte(body, BYTE_NEWLINE, pos + 2) or (body_len + 1)
|
||||
return body:sub(pos + 2, end_pos - 1)
|
||||
end
|
||||
return nil
|
||||
else
|
||||
return nil
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
--- Parse a decimal/negative-decimal/hex integer literal starting at byte position `start`.
|
||||
--- Returns (value, end_pos) on success, or (nil, start) on failure / no match.
|
||||
--- Accepts: 12, -1, 0, 0x10, 0X1F, -0x10.
|
||||
@@ -1128,6 +1172,46 @@ local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
|
||||
return marker_end
|
||||
end
|
||||
|
||||
--- Parse `atom_auto_reg(<atom>, R_<Sym>)` and `phase_auto_reg(<phase>, R_<Sym>)` markers.
|
||||
---
|
||||
--- The macros expand to `sym = sym##_Code` per their definition in dsl.atom.h.
|
||||
--- After preprocessing, the marker renders as a full enum entry of the form `R_<Sym> = R_<Sym>_Code,`.
|
||||
--- This parser detects the macro invocation site, extracts `(scope_name, sym)`, and stores it
|
||||
--- in the per-source table (atom_auto_regs or phase_auto_regs) under the scope's name.
|
||||
---
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @return integer
|
||||
local function parse_auto_reg_marker(source, pos, ident_end, line_of, out)
|
||||
local marker_kind = source:sub(pos, ident_end - 1) -- "atom_auto_reg" or "phase_auto_reg"
|
||||
local scope_kind = marker_kind == "atom_auto_reg" and "atom" or "phase"
|
||||
|
||||
local inner, after_paren = read_parens_after(source, ident_end)
|
||||
if not inner then return after_paren end
|
||||
|
||||
local args = duffle.split_top_level_commas(inner)
|
||||
local scope_name = args[1] and duffle.trim(args[1]) or nil
|
||||
local sym = args[2] and duffle.trim(args[2]) or nil
|
||||
|
||||
-- Filter: only accept `R_<Sym>` form (matches `^R_[%w_]+$`).
|
||||
if scope_name and sym and sym:match("^R_[%w_]+$") then
|
||||
if scope_kind == "atom" then
|
||||
out.atom_auto_regs = out.atom_auto_regs or {}
|
||||
out.atom_auto_regs[scope_name] = out.atom_auto_regs[scope_name] or {}
|
||||
out.atom_auto_regs[scope_name][sym] = sym
|
||||
else
|
||||
out.phase_auto_regs = out.phase_auto_regs or {}
|
||||
out.phase_auto_regs[scope_name] = out.phase_auto_regs[scope_name] or {}
|
||||
out.phase_auto_regs[scope_name][sym] = sym
|
||||
end
|
||||
end
|
||||
|
||||
return after_paren
|
||||
end
|
||||
|
||||
-- Parse `atom_dbg_reg_default(R_X, <type>...)`;
|
||||
-- the second argument may be a `Type` or `Type*`/`Type**` chain. Records in `out.types[R_X]`.
|
||||
local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
|
||||
@@ -1282,6 +1366,49 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
|
||||
return after_paren
|
||||
end
|
||||
|
||||
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
|
||||
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
|
||||
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit
|
||||
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
|
||||
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
|
||||
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
|
||||
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
|
||||
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
|
||||
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @return integer
|
||||
local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
|
||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||
if not inner then return after_paren end
|
||||
|
||||
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
|
||||
local last_brace_pos = nil
|
||||
for search_pos = #inner, 1, -1 do
|
||||
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
|
||||
end
|
||||
if not last_brace_pos then return after_paren end
|
||||
|
||||
-- Use duffle.read_braces to find the matching close brace.
|
||||
-- Uses `read_balanced` for delimiter-depth tracking.
|
||||
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
|
||||
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
||||
if close_pos > #inner + 1 then return after_paren end
|
||||
|
||||
-- The atom name is the FIRST ident of the args (matches MipsAtomComp_Proc_'s "first ident" rule).
|
||||
-- MipsAtom_Proc_ has no `ac_` prefix; `strip_ac_prefix` is a no-op for unprefixed names.
|
||||
local raw_name = inner:match("^%s*([%w_]+)") or "?"
|
||||
local name = strip_ac_prefix(raw_name)
|
||||
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
|
||||
local body_off = open_paren + 2 + last_brace_pos
|
||||
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
||||
|
||||
return after_paren
|
||||
end
|
||||
|
||||
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
@@ -1602,6 +1729,16 @@ local function parse_enum_entry(source, body, body_offset, line_of, out, entry_n
|
||||
local value, value_end = parse_enum_value(body, after_ws, out)
|
||||
if value == nil then return value_start end
|
||||
|
||||
-- Capture the trailing C-comment (if any) before `skip_ws_and_cmt` discards it.
|
||||
-- The `atom_auto_reg(<scope>, <sym>)` macro expands to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
|
||||
-- so the scope name lives in the comment after the RHS value. Routes through `out.atom_entry_comments`
|
||||
-- for downstream `parse_enum` to split into `out.atom_auto_regs` / `out.phase_auto_regs`.
|
||||
local trailing_cmt = read_trailing_cmt_after(body, value_end)
|
||||
if trailing_cmt then
|
||||
out.atom_entry_comments = out.atom_entry_comments or {}
|
||||
out.atom_entry_comments[entry_name] = trailing_cmt
|
||||
end
|
||||
|
||||
local after_value = duffle.skip_ws_and_cmt(body, value_end)
|
||||
local has_atom_reg, end_after_atom_reg = check_bare_atom_reg(body, after_value)
|
||||
|
||||
@@ -1657,15 +1794,24 @@ local function parse_enum_body(source, body, body_offset, line_of, out)
|
||||
else
|
||||
local entry_name, name_end = duffle.read_ident(body, pos)
|
||||
if entry_name then
|
||||
local after_name = duffle.skip_ws_and_cmt(body, name_end)
|
||||
if body:byte(after_name) == BYTE_EQUAL then
|
||||
local new_pos = parse_enum_entry(
|
||||
source, body, body_offset, line_of, out,
|
||||
entry_name, pos, after_name + 1
|
||||
)
|
||||
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
|
||||
-- In-enum `atom_auto_reg(<scope>, R_<Sym>)` / `phase_auto_reg(<scope>, R_<Sym>)` markers:
|
||||
-- the C preprocessor expands them to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
|
||||
-- but the metaprogram reads source-as-written so we must dispatch the parser here too.
|
||||
-- Mirrors the top-level `DECL_PARSERS` entry for `atom_auto_reg` / `phase_auto_reg`.
|
||||
if entry_name == "atom_auto_reg" or entry_name == "phase_auto_reg" then
|
||||
local new_pos = parse_auto_reg_marker(body, pos, name_end, line_of, out)
|
||||
if new_pos > pos then pos = new_pos else pos = name_end end
|
||||
else
|
||||
pos = name_end
|
||||
local after_name = duffle.skip_ws_and_cmt(body, name_end)
|
||||
if body:byte(after_name) == BYTE_EQUAL then
|
||||
local new_pos = parse_enum_entry(
|
||||
source, body, body_offset, line_of, out,
|
||||
entry_name, pos, after_name + 1
|
||||
)
|
||||
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
|
||||
else
|
||||
pos = name_end
|
||||
end
|
||||
end
|
||||
else
|
||||
pos = pos + 1
|
||||
@@ -1695,6 +1841,25 @@ local function parse_enum(source, pos, ident_end, line_of, out)
|
||||
if not body then return after_brace end
|
||||
parse_enum_body(source, body, body_off, line_of, out)
|
||||
|
||||
-- Route `atom_auto_reg:` / `phase_auto_reg:` markers discovered in trailing C-comments
|
||||
-- into the per-source `atom_auto_regs` / `phase_auto_regs` projections.
|
||||
-- Pattern matches the RHS expansion `R_<Sym> = R_<Sym>_Code /* <kind>_auto_reg: <scope> */`
|
||||
-- emitted by the `atom_auto_reg` / `phase_auto_reg` macros in dsl.atom.h.
|
||||
for entry_name, cmt_text in pairs(out.atom_entry_comments or {}) do
|
||||
local atom_scope = cmt_text:match("atom_auto_reg:%s*([%w_]+)")
|
||||
if atom_scope then
|
||||
out.atom_auto_regs = out.atom_auto_regs or {}
|
||||
out.atom_auto_regs[atom_scope] = out.atom_auto_regs[atom_scope] or {}
|
||||
out.atom_auto_regs[atom_scope][entry_name] = entry_name
|
||||
end
|
||||
local phase_scope = cmt_text:match("phase_auto_reg:%s*([%w_]+)")
|
||||
if phase_scope then
|
||||
out.phase_auto_regs = out.phase_auto_regs or {}
|
||||
out.phase_auto_regs[phase_scope] = out.phase_auto_regs[phase_scope] or {}
|
||||
out.phase_auto_regs[phase_scope][entry_name] = entry_name
|
||||
end
|
||||
end
|
||||
|
||||
return after_brace
|
||||
end
|
||||
|
||||
@@ -1708,12 +1873,18 @@ end
|
||||
|
||||
local DECL_PARSERS = {
|
||||
MipsAtom_ = parse_mips_atom,
|
||||
MipsAtom_Proc_ = parse_mips_atom_proc,
|
||||
MipsAtomComp_ = parse_mips_atom_comp,
|
||||
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
|
||||
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
|
||||
-- identifier follows the ordinary unrelated-token path; there is no alias.
|
||||
atom_dbg_skip = parse_dbg_skip_marker,
|
||||
atom_dbg_reg_default = parse_atom_dbg_reg_default,
|
||||
-- `atom_auto_reg(atom, R_<Sym>)` and `phase_auto_reg(phase, R_<Sym>)` populate per-source
|
||||
-- `out.atom_auto_regs` / `out.phase_auto_regs`; the cross-source merge lands in
|
||||
-- `corpus.atom_auto_regs` / `corpus.phase_auto_regs` (first-wins).
|
||||
atom_auto_reg = parse_auto_reg_marker,
|
||||
phase_auto_reg = parse_auto_reg_marker,
|
||||
MipsCode = parse_mips_code,
|
||||
typedef = parse_typedef_binds,
|
||||
_Pragma = parse_pragma_macro,
|
||||
@@ -1748,6 +1919,14 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
||||
debug_skip_markers = {},
|
||||
types = {},
|
||||
atom_views = {},
|
||||
-- Per-source projection for `atom_auto_reg(<atom>, R_<Sym>)` markers.
|
||||
-- Each entry is keyed by atom_name; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
|
||||
-- Merged cross-source into `corpus.atom_auto_regs` (first-wins).
|
||||
atom_auto_regs = {},
|
||||
-- Per-source projection for `phase_auto_reg(<phase>, R_<Sym>)` markers.
|
||||
-- Each entry is keyed by phase_label; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
|
||||
-- Merged cross-source into `corpus.phase_auto_regs` (first-wins).
|
||||
phase_auto_regs = {},
|
||||
line_of = line_of,
|
||||
-- Source-derived register-alias registry (atom_reg opt-in entries).
|
||||
-- Keys are full R_* idents (never stripped); see parse_enum / parse_enum_body.
|
||||
@@ -1987,6 +2166,8 @@ local function merge_corpus_registries(corpus)
|
||||
corpus.atom_ctxs = corpus.atom_ctxs or {}
|
||||
corpus.atom_phases = corpus.atom_phases or {}
|
||||
corpus.atom_infos = corpus.atom_infos or {}
|
||||
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
|
||||
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
|
||||
corpus.collisions = corpus.collisions or {}
|
||||
|
||||
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
||||
@@ -2030,7 +2211,7 @@ local function merge_corpus_registries(corpus)
|
||||
corpus.collisions, "binds", bind_shape)
|
||||
end
|
||||
|
||||
-- atoms_by_name: MipsAtom_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
|
||||
-- atoms_by_name: MipsAtom_(name) + MipsAtom_Proc_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
|
||||
-- Each atom carries `{line, name, body, body_off, kind, raw_name, ...}`.
|
||||
-- Duplicate atom names across sources are first-wins + collision; see the atom_infos block below for the evidence list.
|
||||
for _, atom_entry in ipairs(scan.atoms or {}) do
|
||||
@@ -2065,6 +2246,22 @@ local function merge_corpus_registries(corpus)
|
||||
corpus.collisions, "phase", phase_shape)
|
||||
end
|
||||
|
||||
-- atom_auto_regs: keyed by atom scope name; each carries a `{R_<Sym> = R_<Sym>}` map.
|
||||
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
|
||||
for atom_scope, syms in pairs(scan.atom_auto_regs or {}) do
|
||||
if corpus.atom_auto_regs[atom_scope] == nil then
|
||||
corpus.atom_auto_regs[atom_scope] = syms
|
||||
end
|
||||
end
|
||||
|
||||
-- phase_auto_regs: keyed by phase label; each carries a `{R_<Sym> = R_<Sym>}` map.
|
||||
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
|
||||
for phase_label, syms in pairs(scan.phase_auto_regs or {}) do
|
||||
if corpus.phase_auto_regs[phase_label] == nil then
|
||||
corpus.phase_auto_regs[phase_label] = syms
|
||||
end
|
||||
end
|
||||
|
||||
-- atom_infos: ALWAYS append every record in source/declaration order.
|
||||
-- Duplicates are preserved so the annotation pass can flag them via `check_unique_annotation`;
|
||||
-- The merge is purely order-preserving.
|
||||
|
||||
@@ -256,8 +256,21 @@ local BRANCH_PATTERN = "^branch_[%w_]+%s*%("
|
||||
-- The C preprocessor expands it BEFORE the metaprogram sees the source, but for source-level metadata consistency we still match it here and classify it as a branch_equal.
|
||||
-- This keeps `consuming_encoder` canonical for any downstream tooling that consults the metadata field.
|
||||
local JUMP_REL_PATTERN = "^jump_rel%s*%("
|
||||
local UNCOND_JUMP_PATTERN = "^%f[%w](jump|call_addr)%f[%W]"
|
||||
local TERMINAL_JUMP_PATTERN = "^%f[%w](jump_reg|call_reg|jump_link)%f[%W]"
|
||||
local UNCOND_JUMP_PATTERNS = {
|
||||
"^%f[%w]jump%f[%W]",
|
||||
"^%f[%w]call_addr%f[%W]",
|
||||
}
|
||||
local TERMINAL_JUMP_PATTERNS = {
|
||||
"^%f[%w]jump_reg%f[%W]",
|
||||
"^%f[%w]call_reg%f[%W]",
|
||||
"^%f[%w]jump_link%f[%W]",
|
||||
}
|
||||
local function matches_any(tok, patterns)
|
||||
for i = 1, #patterns do
|
||||
if tok:match(patterns[i]) then return true end
|
||||
end
|
||||
return false
|
||||
end
|
||||
|
||||
local function classify_tokens(tokens)
|
||||
local n = #tokens
|
||||
@@ -301,13 +314,13 @@ local function classify_tokens(tokens)
|
||||
-- Both encode a 16-bit signed relative word offset.
|
||||
is_branch = true
|
||||
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
|
||||
elseif tok:match(UNCOND_JUMP_PATTERN) then
|
||||
elseif matches_any(tok, UNCOND_JUMP_PATTERNS) then
|
||||
-- Unconditional absolute jump / call: `jump(off)` / `call_addr(off)`.
|
||||
-- One immediate offset field; can carry an `atom_offset(F, T)` marker (the offsets pass dispatches on `consuming_encoder` — see `passes/offsets.lua::compute_offsets`).
|
||||
is_branch = true
|
||||
is_unconditional_jump = true
|
||||
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
|
||||
elseif tok:match(TERMINAL_JUMP_PATTERN) then
|
||||
elseif matches_any(tok, TERMINAL_JUMP_PATTERNS) then
|
||||
-- Register-form jump / call: no offset field; `atom_offset` is invalid here (the offsets pass will error if one is supplied).
|
||||
-- Transfers control OUT of the current atom — the CFG treats this as a path terminator.
|
||||
is_terminal_jump = true
|
||||
@@ -567,13 +580,31 @@ local function evaluate_gpr_value_rule(rule, ev_args, gpr_values)
|
||||
return shift_left_u4(immediate % 0x10000, 16)
|
||||
end
|
||||
|
||||
local source = nil
|
||||
-- Encoders that take `R_0` implicitly (e.g. `li_s(rt, imm)` which is `add_ui(rt, R_0, imm)`) have a non-GPR operand at the source position.
|
||||
-- Fall back to R_0 = 0.
|
||||
-- The implicit-R_0 macros also use a different immediate position (e.g. `li_s`'s `add_ui` rule has source = 2 / immediate = 3
|
||||
-- but the macro takes 2 args); when the configured immediate position is out of bounds.
|
||||
-- Fall back instead to scanning the macro's args for the first integer literal and use that as the immediate.
|
||||
local source = 0
|
||||
if rule.source then
|
||||
source = constant_for_operand(gpr_values, ev_args[rule.source])
|
||||
if source == nil then return nil end
|
||||
if is_gpr_operand(ev_args[rule.source]) then
|
||||
source = constant_for_operand(gpr_values, ev_args[rule.source])
|
||||
if source == nil then return nil end
|
||||
end
|
||||
-- Non-GPR at source position = implicit R_0; source stays 0.
|
||||
end
|
||||
local immediate = nil
|
||||
if rule.immediate and ev_args[rule.immediate] ~= nil then
|
||||
immediate = parse_integer_literal(ev_args[rule.immediate])
|
||||
if immediate == nil then return nil end
|
||||
elseif rule.immediate then
|
||||
-- Immediate position out of bounds: scan for the first integer literal in the args.
|
||||
for _, arg in ipairs(ev_args) do
|
||||
immediate = parse_integer_literal(arg)
|
||||
if immediate ~= nil then break end
|
||||
end
|
||||
if immediate == nil then return nil end
|
||||
end
|
||||
local immediate = rule.immediate and parse_integer_literal(ev_args[rule.immediate]) or nil
|
||||
if rule.immediate and immediate == nil then return nil end
|
||||
if operation == "add_ui" then return wrap_u4( source + sign_extend_i16(immediate))
|
||||
elseif operation == "or_i" then return bit_binary( source, immediate % 0x10000, "or")
|
||||
elseif operation == "and_i" then return bit_binary( source, immediate % 0x10000, "and")
|
||||
@@ -1433,17 +1464,21 @@ end
|
||||
--- The register becomes non-volatile again at word N+2 (the load has retired), OR sooner if a non-load instruction overwrites the register
|
||||
--- (the overwriter's write is the fresh producer; the load's value is shadowed and never observed by any reader).
|
||||
---
|
||||
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt: their internal load-then-use sequences
|
||||
--- are part of the fixed handshake (e.g. `ac_load_tri_indices` loads into R_T0..R_T2, but those are caller-supplied).
|
||||
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt from some checks, but load-delay
|
||||
--- safety applies to their emitted instructions as well.
|
||||
---
|
||||
--- The walker reads `duffle.OPERAND_READ_POSITIONS[event.encoder]` to determine which args are read-source
|
||||
--- (the destination of a load is in `writes`, not `reads` — see `duffle.INSTRUCTION_GPR_EFFECTS`).
|
||||
--- The check is purely structural; it does not consult the GPR-value lattice (no constant propagation needed for load-delay detection — the volatility window is unconditional).
|
||||
--- The check is purely structural; it does not consult the GPR-value lattice
|
||||
--- (no constant propagation needed for load-delay detection — the volatility window is unconditional).
|
||||
local function check_load_delay_slots(atom, pipe_ctx, findings)
|
||||
if atom.kind ~= "atom" then return end
|
||||
local events = atom.paths.word_events or {}
|
||||
-- The load-delay check applies to every atom and component body, including debug-skipped components (`ac_*` and `atom_dbg_skip MipsAtom_(...)`).
|
||||
-- The `atom_dbg_skip` marker controls debugger stepping, not instruction safety.
|
||||
-- `atom_proc` atoms have full bodies with loads that need delay slots, so the check applies to them too.
|
||||
local p = atom.paths or {}
|
||||
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||
local events = p.word_events or {}
|
||||
if #events == 0 then return end
|
||||
if is_runtime_helper(atom) then return end
|
||||
|
||||
local gpr_effects = duffle.INSTRUCTION_GPR_EFFECTS or {}
|
||||
local read_positions = duffle.OPERAND_READ_POSITIONS or {}
|
||||
@@ -1545,6 +1580,8 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
||||
if is_runtime_helper(atom) then return end
|
||||
-- Per-kind semantics:
|
||||
-- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job.
|
||||
-- MipsAtom_Proc_ (runtime-proc atom): exactly 1 mac_yield at the end of the body. Same as baked atom;
|
||||
-- the proc IS the atom; the runtime call to `atombuilder_unroll` doesn't introduce a parent atom.
|
||||
-- MipsAtomComp_ (bare static-array component): ZERO mac_yield.
|
||||
-- The component is invoked from inside an atom body; the parent atom does the yield.
|
||||
-- MipsAtomComp_Proc_ (procedural component): ZERO mac_yield.
|
||||
@@ -1568,7 +1605,7 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
||||
return atom.line + line_in_body[tokens[idx].rel]
|
||||
end
|
||||
|
||||
if atom.kind == "atom" then
|
||||
if atom.kind == "atom" or atom.kind == "atom_proc" then
|
||||
-- Baked atom: exactly 1 yield at the end.
|
||||
if count == 0 then
|
||||
findings[#findings + 1] = {
|
||||
@@ -1613,6 +1650,7 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
||||
-- The parent atom does the yield.
|
||||
-- A yield inside a component would either be dead code (bare) or prematurely terminate the function (proc).
|
||||
-- Both are bugs.
|
||||
-- `atom_proc` atoms are NOT components; they're runtime-proc atoms that own their own yield (handled in the `if` branch above).
|
||||
if count > 0 then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
@@ -1644,7 +1682,7 @@ end
|
||||
--- Per-atom. Runtime-helper atoms (`debug_skip`) are exempt.
|
||||
--- Takes `(atom, pipe_ctx, findings)`; `pipe_ctx` is unused.
|
||||
local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
||||
if atom.kind ~= "atom" then return end
|
||||
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||
if is_runtime_helper(atom) then return end
|
||||
|
||||
local tokens = atom.paths.tokens
|
||||
@@ -1656,21 +1694,39 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
||||
return atom.line + line_in_body[tokens[idx].rel]
|
||||
end
|
||||
|
||||
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot.
|
||||
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s (natural fall-through load pattern).
|
||||
-- When the pattern is satisfied, the check stays silent; only violations emit findings.
|
||||
for tok_idx = 1, n do
|
||||
local c = tc[tok_idx]
|
||||
if c.ident == "mac_yield_load" then
|
||||
if tok_idx < 2 or not tc[tok_idx - 1].is_branch then
|
||||
local prev_ident = (tok_idx >= 2) and (tc[tok_idx - 1].ident or "?") or "<none>"
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
|
||||
check = "yield_load_tail_pairing",
|
||||
kind = "error",
|
||||
msg = string.format(
|
||||
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — `mac_yield_load()` must fill a branch BD-slot."
|
||||
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident),
|
||||
}
|
||||
local prev_tc = (tok_idx >= 2) and tc[tok_idx - 1] or nil
|
||||
-- Look for the next `atom_label()` token (skip `atom_offset` markers; check immediately-adjacent first).
|
||||
local next_label_tc = (tok_idx + 1 <= n) and tc[tok_idx + 1] or nil
|
||||
if next_label_tc and next_label_tc.ident ~= "atom_label" then
|
||||
next_label_tc = nil
|
||||
for j = tok_idx + 1, n do
|
||||
local t = tc[j]
|
||||
if t.ident == "atom_label" then
|
||||
next_label_tc = t
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
local natural_fallthrough = prev_tc and prev_tc.is_atom_label and next_label_tc ~= nil
|
||||
if not natural_fallthrough then
|
||||
if tok_idx < 2 or not prev_tc.is_branch then
|
||||
local prev_ident = prev_tc and (prev_tc.ident or "?") or "<none>"
|
||||
local next_ident = next_label_tc and (next_label_tc.ident .. "(" .. (next_label_tc.label_name or "?") .. ")") or "<no following label>"
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
|
||||
check = "yield_load_tail_pairing",
|
||||
kind = "error",
|
||||
msg = string.format(
|
||||
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — and the next `atom_label()` token is `%s` — `mac_yield_load()` must fill a branch BD-slot or sit between two `atom_label`s for the natural fall-through load."
|
||||
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -1845,9 +1901,9 @@ end
|
||||
--- - Atoms containing a `mac_<name>(...)` call whose `name` is not registered in `pipe_ctx.components_by_name` emit a "new macro;
|
||||
--- Not in corpus.components" advisory — the auto-derivation returned nil for that name.
|
||||
---
|
||||
--- Applies only to `kind = "atom"` (baked atoms). Components don't emit full primitives.
|
||||
--- Applies only to `kind = "atom"` or `kind = "atom_proc"` (full-atom bodies). Components don't emit full primitives.
|
||||
local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
|
||||
if atom.kind ~= "atom" then return end
|
||||
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||
local tokens = atom.paths.tokens
|
||||
local line_in_body = atom.paths.line_in_body
|
||||
local tc = atom.paths.tok_class
|
||||
@@ -2015,8 +2071,9 @@ local function analyze_atom_paths(atom, pipe_ctx)
|
||||
succ[#succ + 1] = label_pos + 1
|
||||
end
|
||||
end
|
||||
-- For literal-offset jumps (label == false), the target is a non-tracked address; conservatively omit.
|
||||
return succ, nil
|
||||
-- For literal-offset jumps (label == false), control transfers out unconditionally.
|
||||
-- Treat as a terminator so the path is recorded (NOT as a silent fall-through to the next token, which is unreachable in this atom's execution).
|
||||
return {}, tok_idx
|
||||
end
|
||||
-- Conditional branch: BD slot absorbed; two successors — fall-through (tok_idx+2) + taken (if known).
|
||||
if tok_idx + 2 <= n then
|
||||
@@ -2032,9 +2089,11 @@ local function analyze_atom_paths(atom, pipe_ctx)
|
||||
-- Return (succ, nil), the second value is the terminator marker (nil = not a terminator).
|
||||
return succ, nil
|
||||
end
|
||||
-- Normal token: just the next one
|
||||
-- Normal token: just the next one.
|
||||
-- The final ordinary word of the body has no successor and terminates the path;
|
||||
-- record it as an implicit endpoint so the cycle budget for non-yield components is not silently zeroed.
|
||||
if tok_idx + 1 <= n then return { tok_idx + 1 }, nil end
|
||||
return {}, nil
|
||||
return {}, tok_idx
|
||||
end
|
||||
|
||||
-- DFS through all paths. Track the current cycle sum, a visited set scoped to the current path (to detect loops), and a count of paths.
|
||||
|
||||
@@ -118,6 +118,12 @@ local PASSES = {
|
||||
kind = "header-output",
|
||||
deps = {"scan-source", "word-counts"},
|
||||
},
|
||||
auto_reg = {
|
||||
module = "passes.auto_reg",
|
||||
kind = "header-output",
|
||||
deps = {"components"},
|
||||
groups = { "pre-link" },
|
||||
},
|
||||
["emission-model"] = {
|
||||
module = "passes.emission_model",
|
||||
kind = "validation",
|
||||
|
||||
@@ -14,16 +14,21 @@ $url_armips = 'https://github.com/Kingcom/armips.git'
|
||||
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
||||
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
||||
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
||||
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
|
||||
|
||||
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
|
||||
|
||||
$path_armips = join-path $path_toolchain 'armips'
|
||||
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||
$path_lpeg = join-path $path_toolchain 'lpeg'
|
||||
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
|
||||
|
||||
clone-gitrepo $path_armips $url_armips
|
||||
clone-gitrepo $path_lpeg $url_lpeg
|
||||
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
||||
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
||||
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
|
||||
|
||||
$path_armips_build = join-path $path_armips 'build'
|
||||
verify-path $path_armips_build
|
||||
|
||||
Reference in New Issue
Block a user