30 changed files with 1925 additions and 610 deletions
+1
View File
@@ -20,3 +20,4 @@ toolchain/lpeg
scratch scratch
toolchain/libpsn00b toolchain/libpsn00b
scripts/pcsx_debug_helper.zip
+14
View File
@@ -0,0 +1,14 @@
#ifdef INTELLISENSE_DIRECTIVES
# pragma once
#endif
enum {
bios_init_pad_2 = 0x12,
bios_start_pad_2 = 0x13,
bios_flushcache = 0x44,
bios_table_addr = 0xA0,
bios_btable_addr = 0xB0,
};
enum {
bios_pad_buffer_size = 0x22,
};
+7 -3
View File
@@ -28,8 +28,9 @@
#define internal static // internal #define internal static // internal
#define asm __asm__ #define asm __asm__
#define align_(value) __attribute__((aligned (value))) // for easy alignment
#define A_(data) (& data)
#define align_(value) __attribute__((aligned (value))) // for easy alignment
#define align_(value) __attribute__((aligned (value))) // for easy alignment #define align_(value) __attribute__((aligned (value))) // for easy alignment
#define C_(type,data) ((type)(data)) // for enforced precedence #define C_(type,data) ((type)(data)) // for enforced precedence
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path #define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
@@ -133,8 +134,8 @@ typedef __UINT32_TYPE__ TSet_(B4);
#define u4_v(value) C_(U4 V_*, value) #define u4_v(value) C_(U4 V_*, value)
enum { false = 0, true = 1, true_overflow, }; enum { false = 0, true = 1, true_overflow, };
#define u4_lo(value) ((value) & 0xFFFFU) #define u4_lo(value) (u4_(value) & 0xFFFFU)
#define u4_hi(value) ((value) >> 12) #define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
typedef void Proc_(VoidFn) (void); typedef void Proc_(VoidFn) (void);
@@ -168,6 +169,8 @@ def_signed_ops(le, <=)
#undef def_signed_ops #undef def_signed_ops
#undef def_signed_op #undef def_signed_op
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
#if 0
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__) #define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
#define add_s(a,b) def_generic_sop(add,a,b) #define add_s(a,b) def_generic_sop(add,a,b)
#define sub_s(a,b) def_generic_sop(sub,a,b) #define sub_s(a,b) def_generic_sop(sub,a,b)
@@ -177,6 +180,7 @@ def_signed_ops(le, <=)
#define ge_s(a,b) def_generic_sop(ge, a,b) #define ge_s(a,b) def_generic_sop(ge, a,b)
#define le_s(a,b) def_generic_sop(le, a,b) #define le_s(a,b) def_generic_sop(le, a,b)
#undef def_generic_sop #undef def_generic_sop
#endif
#define alignas _Alignas #define alignas _Alignas
#define alignof _Alignof #define alignof _Alignof
+130 -15
View File
@@ -14,7 +14,9 @@
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h // source: C:\projects\Pikuma\ps1\code\duffle\pad.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h // source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h // source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h // source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
@@ -68,6 +70,27 @@ WORD_COUNT(mac_load_v2s2, 2)
, store_half(rt_y, base, offset + O_(V2_S2,y)) , store_half(rt_y, base, offset + O_(V2_S2,y))
WORD_COUNT(mac_store_v2s2, 2) WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
load_word( rs_x, r_base, O_(V3_S4,x)) \
, load_word( rs_y, r_base, O_(V3_S4,y)) \
, load_word( rs_z, r_base, O_(V3_S4,z))
WORD_COUNT(mac_load_v3s4, 3)
/* atom_dbg_skip */
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
store_word(rt_x, base, offset + O_(V3_S4,x)) \
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
, store_word(rt_z, base, offset + O_(V3_S4,z))
WORD_COUNT(mac_store_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
sub_s(rds_x, rds_x, rt_x) \
, sub_s(rds_y, rds_y, rt_y) \
, sub_s(rds_z, rds_z, rt_z)
WORD_COUNT(mac_sub_v3s4, 3)
/* atom_dbg_skip */ /* atom_dbg_skip */
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \ #define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
store_half(rt_x, base, offset + O_(Rect_S2,x)) \ store_half(rt_x, base, offset + O_(Rect_S2,x)) \
@@ -124,6 +147,86 @@ WORD_COUNT(mac_gte_store_g4_p012, 3)
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
WORD_COUNT(mac_gte_store_g4_p3, 1) WORD_COUNT(mac_gte_store_g4_p3, 1)
/* atom_dbg_skip */
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop \
, gte_cmdw_sqr \
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
WORD_COUNT(mac_gte_sqr_v3, 8)
/* atom_dbg_skip */
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
gte_mv_to_data_r(r_recip_est, C2_IR0) \
, gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_dx, C2_MAC1) \
, gte_mv_from_data_r(r_dy, C2_MAC2) \
, gte_mv_from_data_r(r_dz, C2_MAC3) \
, shift_aright_var(r_dx, r_dx, r_shift) \
, shift_aright_var(r_dy, r_dy, r_shift) \
, shift_aright_var(r_dz, r_dz, r_shift)
WORD_COUNT(mac_gte_gpf_scale, 13)
/* atom_dbg_skip */
#define mac_normalize_v3s4(r_sx, r_sy, r_sz, r_sq_y, r_sq_z, r_recip_est, r_lzcr, r_shift, r_tmp) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop \
, gte_cmdw_sqr /* ─── Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS ─── // Note: r_recip_est first used as the sum accumulator (= |v|²), which is also what LZCS needs. */ \
, gte_mv_from_data_r(r_sq_y, C2_MAC1) /* r_sq_y = MAC1 = sx² */ \
, gte_mv_from_data_r(r_sq_z, C2_MAC2) /* r_sq_z = MAC2 = sy² */ \
, gte_mv_from_data_r(r_recip_est, C2_MAC3) /* r_recip_est = MAC3 = sz² */ \
, nop /* MFC2→GPR load delay (1 slot) */ \
, add_u(r_recip_est, r_recip_est, r_sq_z) /* r_recip_est += sy² */ \
, add_u(r_recip_est, r_recip_est, r_sq_y) /* r_recip_est += sx² (sum = |v|²) */ \
, gte_mv_to_data_r( r_recip_est, C2_LZCS) /* LZCS = |v|² */ \
, nop2 \
, gte_mv_from_data_r(r_lzcr, C2_LZCR) /* r_lzcr = LZCR (count of leading bits) */ \
, nop /* MFC2→GPR load delay (1 slot) */ /* ─── Stage 3: compute shift amount, align |v|² to bit 24, lookup 1/|v| ─── // Matches libgte `bltz +0x10 ; nop ; b +0x14 ; sllv t4,v0,t3` pattern: // - bltz TAKEN → nop (BD), jump to srav_path; sllv SKIPPED // - bltz !TAKEN → nop (BD), b +0x14 jumps to aligned_done; sllv (BD of b) executes */ \
, and_i( r_lzcr, r_lzcr, -2) /* r_lzcr &= ~1 (force even for halving) */ \
, li_s( r_shift, 31) /* r_shift = 31 */ \
, sub_s( r_shift, r_shift, r_lzcr) /* r_shift = 31 - LZCR */ \
, shift_aright( r_shift, r_shift, 1) /* r_shift = (31 - LZCR) / 2 */ \
, add_si( r_tmp, r_lzcr, -24) /* r_tmp = LZCR - 24 (signed, for branch) */ \
, branch_lt_zero(r_tmp, atom_offset(srav_path, aligned_done)) \
, nop \
, jump_rel( atom_offset(aligned_done, srav_path)) \
, shift_lleft_var(r_recip_est, r_recip_est, r_tmp) /* BD-slot of branch_equal: r_recip_est = |v|² << (LZCR - 24) */ \
, atom_label(srav_path) /* SRAV path: |v|² is small (top bit < bit 24) */ \
, li_s( r_tmp, 24) \
, sub_s( r_tmp, r_tmp, r_lzcr) /* r_tmp = 24 - LZCR */ \
, shift_aright_var(r_recip_est, r_recip_est, r_tmp) /* r_recip_est = |v|² >> (24 - LZCR) */ \
, atom_label(aligned_done) /* Both paths converge here with |v|² aligned to bit 24 */ /* r_recip_est now holds |v|² aligned to bit 24 — convert to byte offset, -64 to skip zero pad. */ \
, add_si( r_recip_est, r_recip_est, -64) \
, shift_lleft( r_recip_est, r_recip_est, 1) /* r_recip_est *= 2 (half-word index) */ /* Reference OUR local sqrtbl via &-address split. Compiler/linker resolves both halves. */ \
, load_upper_i( r_tmp, u4_hi(& gte_normalize_sqr_tbl)) /* lui */ \
, or_i_self( r_tmp, u4_lo(& gte_normalize_sqr_tbl)) /* ori */ \
, add_u( r_tmp, r_tmp, r_recip_est) /* r_tmp = sqrtbl base + byte offset (matches libgte 0x80016118: addu t5,t5,t4) */ \
, load_half( r_recip_est, r_tmp, 0) /* r_recip_est = sqrtbl[r_recip_est] = 1/|v| estimate */ \
, nop /* retire load_half before MTC2 (matches libgte 0x80016120: nop) */ /* ─── Stage 4: mtc2 IR0..3, GPF (MAC = IR0*IR), mfc2 MAC, srav finalize ─── // Componentized equivalent: mac_gte_gpf_scale. */ \
, gte_mv_to_data_r(r_recip_est, C2_IR0) /* IR0 = 1/|v| estimate */ \
, gte_mv_to_data_r(r_sx, C2_IR1) /* IR1 = src.x */ \
, gte_mv_to_data_r(r_sy, C2_IR2) /* IR2 = src.y */ \
, gte_mv_to_data_r(r_sz, C2_IR3) /* IR3 = src.z */ \
, nop2 /* COP2 transfer latency (2 slots) */ \
, gte_cmdw_gpf \
, gte_mv_from_data_r(r_sx, C2_MAC1) /* MAC1 → r_sx (overwrites src.x with raw reciprocal-scaled) */ \
, gte_mv_from_data_r(r_sy, C2_MAC2) \
, gte_mv_from_data_r(r_sz, C2_MAC3) \
, shift_aright_var(r_sx, r_sx, r_shift) \
, shift_aright_var(r_sy, r_sy, r_shift) \
, shift_aright_var(r_sz, r_sz, r_shift)
WORD_COUNT(mac_normalize_v3s4, 48)
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \ #define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
load_upper_i(reg_transfer, cmd >> 16) \ load_upper_i(reg_transfer, cmd >> 16) \
, or_i_self( reg_transfer, cmd & 0xFFFF) \ , or_i_self( reg_transfer, cmd & 0xFFFF) \
@@ -156,29 +259,41 @@ WORD_COUNT(mac_format_f3_color, 3)
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3) , mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
WORD_COUNT(mac_format_g4_color, 12) WORD_COUNT(mac_format_g4_color, 12)
#define mac_insert_ot_tag_f3(r_ot_base, r_prim_cursor) \ #define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \ shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \ , add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \ , load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \ , load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \ , mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \ , or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \ , store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \ , shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \ , shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */ , store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
WORD_COUNT(mac_insert_ot_tag_f3, 11) WORD_COUNT(mac_insert_ot_tag, 11)
#define mac_insert_ot_tag_g4(r_ot_base, r_prim_cursor) \ /* atom_dbg_skip */
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \ #define mac_pad_set_centered_axes(r_state, r_scratch) \
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \ load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF) \
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \ , or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF) \
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \ , store_word( r_scratch, r_state, O_(PadState,axes))
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \ WORD_COUNT(mac_pad_set_centered_axes, 3)
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \ /* atom_dbg_skip */
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \ #define mac_pad_set_id_byte(r_state, r_id, id_value) \
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \ add_ui( r_id, R_0, id_value) \
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */ , store_byte(r_id, r_state, O_(PadState,id))
WORD_COUNT(mac_insert_ot_tag_g4, 11) WORD_COUNT(mac_pad_set_id_byte, 2)
/* atom_dbg_skip */
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
add_ui( r_tmp, R_0, pad_status) \
, store_word(r_tmp, r_state, O_(PadState,status))
WORD_COUNT(mac_pad_set_status, 2)
/* atom_dbg_skip */
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
nor_u( r_buttons, r_buttons, R_0) \
, store_half( r_buttons, r_pad_state, O_(PadState, buttons))
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
+22 -10
View File
@@ -11,7 +11,9 @@
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h // source: C:\projects\Pikuma\ps1\code\duffle\pad.h
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h // source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h // source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h // source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c // source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
@@ -23,17 +25,27 @@
#pragma region duffle #pragma region duffle
// --- atom: pad_bios_snapshot (78 words) --- // --- atom: ac_normalize_v3s4 (48 words) ---
#define _atom_offset_snap_root_skip_disconnected 8 #define _atom_offset_srav_path_aligned_done 6
#define _atom_offset_disconnected_snap_end 61 #define _atom_offset_aligned_done_srav_path 1
#define _atom_offset_case_2_id_dispatch 8
#define _atom_offset_pending_snap_end 51 enum {
#define _atom_offset_id_dispatch_try_analog_stick 11 atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
#define _atom_offset_id_dispatch_snap_end 38 atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
#define _atom_offset_try_analog_stick_try_analog_pad 12 };
#define _atom_offset_analog_stick_snap_end 24
#define _atom_offset_try_analog_pad_try_unsupported 11 // --- atom: pad_bios_snapshot (84 words) ---
#define _atom_offset_snap_root_skip_disconnected 10
#define _atom_offset_disconnected_snap_end 65
#define _atom_offset_case_2_id_dispatch 9
#define _atom_offset_pending_snap_end 54
#define _atom_offset_id_dispatch_try_analog_stick 12
#define _atom_offset_id_dispatch_snap_end 40
#define _atom_offset_try_analog_stick_try_analog_pad 13
#define _atom_offset_analog_stick_snap_end 25
#define _atom_offset_try_analog_pad_try_unsupported 12
#define _atom_offset_analog_pad_snap_end 10 #define _atom_offset_analog_pad_snap_end 10
enum { enum {
+3 -25
View File
@@ -21,8 +21,6 @@ FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_d
store_byte(rb, base, offset + O_(RGB8,b)), store_byte(rb, base, offset + O_(RGB8,b)),
}) })
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
* byte offset. Internal helper used by the *_format_*_color macros. */
FI_ Slice_MipsCode ac_pack_color_word(U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b) FI_ Slice_MipsCode ac_pack_color_word(U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, { atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
load_upper_i(R_AT, (cmd) << 8 | (b)), load_upper_i(R_AT, (cmd) << 8 | (b)),
@@ -30,13 +28,9 @@ atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
store_word( R_AT, r_base, (off)), store_word( R_AT, r_base, (off)),
}) })
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
FI_ Slice_MipsCode ac_format_f3_color(U4 r_base, U1 r, U1 g, U1 b) FI_ Slice_MipsCode ac_format_f3_color(U4 r_base, U1 r, U1 g, U1 b)
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) }) atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
FI_ Slice_MipsCode ac_format_g4_color(U4 r_prim_cursor, FI_ Slice_MipsCode ac_format_g4_color(U4 r_prim_cursor,
U1 r0, U1 g0, U1 b0, U1 r0, U1 g0, U1 b0,
U1 r1, U1 g1, U1 b1, U1 r1, U1 g1, U1 b1,
@@ -49,28 +43,12 @@ MipsAtomComp_Proc_(ac_format_g4_color, {
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3), mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
}) })
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. /* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */ I_ Slice_MipsCode ac_insert_ot_tag(U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, {
I_ Slice_MipsCode ac_insert_ot_tag_f3(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_f3, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1) shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ] add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24 load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
})
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
I_ Slice_MipsCode ac_insert_ot_tag_g4(U4 r_ot_base, U4 r_prim_cursor) MipsAtomComp_Proc_(ac_insert_ot_tag_g4, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24 mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
or_u( R_AT, R_AT, R_V0), // Merge length or_u( R_AT, R_AT, R_V0), // Merge length
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr) store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
+214 -6
View File
@@ -49,20 +49,228 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(U4 r_primitive_cursor) atom_dbg_skip Mip
*/ */
FI_ Slice_MipsCode ac_gte_store_g4_p3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) }) FI_ Slice_MipsCode ac_gte_store_g4_p3(U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
* Stage 2 of normalize consumes these directly.
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
FI_ Slice_MipsCode ac_gte_sqr_v3(U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop, gte_cmdw_sqr,
gte_mv_from_data_r(r_sq_x, C2_MAC1),
gte_mv_from_data_r(r_sq_y, C2_MAC2),
gte_mv_from_data_r(r_sq_z, C2_MAC3),
})
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
* Used standalone for "scale vector by scalar".
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
FI_ Slice_MipsCode ac_gte_gpf_scale(U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, {
gte_mv_to_data_r(r_recip_est, C2_IR0),
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
gte_cmdw_gpf,
gte_mv_from_data_r(r_dx, C2_MAC1),
gte_mv_from_data_r(r_dy, C2_MAC2),
gte_mv_from_data_r(r_dz, C2_MAC3),
shift_aright_var(r_dx, r_dx, r_shift),
shift_aright_var(r_dy, r_dy, r_shift),
shift_aright_var(r_dz, r_dz, r_shift),
})
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
*
* Data is identical to the libgte original (byte-for-byte verified).
*
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the
* mantissa (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
* Sampling the first value of each octave:
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
*
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
* and the load upper_halves of the table bracket the input range.
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
*
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
};
/* ─── Full normalize (all 4 stages inline) ───
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
*
* Component variants that could apply:
* - `ac_gte_sqr_v3` (line ~56) covers stage 1's `mtc2 IR1/2/3 + nop + gte_cmdw_sqr`.
* We do NOT call it because the inlined version of stage 1 is followed immediately by stage 2's `mfc2 MAC1/2/3` chain
* (the operands of `ac_gte_sqr_v3`'s r_sq_x/r_sq_y/r_sq_z would each require an explicit GPR to receive the MAC result,
* then a move to land in r_recip_est for the partial-sum chain).
* Inlining saves ~3 cycles of `or`-merge + register pressure
* (squared MAC3 lands DIRECTLY in r_recip_est which doubles as the partial-sum accumulator and the LZCS input — see r_recip_est row below).
* - `ac_gte_gpf_scale` (line ~71) covers stage 4's `mtc2 IR0..3 + nop2 + gte_cmdw_gpf + mfc2 MAC1/2/3 + sra`.
* We do NOT call it for the symmetric reason: the normalize in-place semantics overwrite the input regs (r_sx/r_sy/r_sz) with the normalized output,
* which `ac_gte_gpf_scale`'s r_dx/r_dy/r_dz output GPRs would not match.
* `gte_cmdw_sqr` and `gte_cmdw_gpf` primitive macros ARE used in the inlined body, so changes to those primitives
* (e.g., the libgte `fake_cmd` signature bits) propagate automatically. The components remain available for callers that want the explicit GPR-shape variants.
*
* Argument aliasing (9 unique physical regs needed, can drop to 8 with r_sq_y ≡ r_lzcr):
* r_sx, r_sy, r_sz : src components in regs (clobbered by mtc2 → IR1/2/3 in stage 1, then by mfc2 MAC1/2/3 in stage 4 — in-place semantics)
* r_sq_y, r_sq_z : MAC2, MAC3 → DIE after stage 2 accumulate (r_sq_y can alias r_lzcr after stage 2 to save one reg)
* r_recip_est : ≡ r_sqmag — multi-purpose (holds |v|² in stage 2, shift-input in stage 3, sqrtbl[index] in stage 4)
* r_lzcr : LZCR value, alive across stage 3 (srav path needs `24 - LZCR`)
* r_shift : (31 - LZCR & ~1) >> 1 — final srav amount (stages 3-4)
* r_tmp : scratch (shift count, branch target, lookup addr, table base)
*
* GPR ccount peak: 9.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
* Words: ~35 (pending re-gen; matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local. */
I_ Slice_MipsCode ac_normalize_v3s4(U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_y, U4 r_sq_z, U4 r_recip_est, U4 r_lzcr, U4 r_shift, U4 r_tmp)
atom_dbg_skip MipsAtomComp_Proc_(ac_normalize_v3s4, {
/* 9-arg signature — must be on one line so the metaprogram captures the full arg list.
* r_sx, r_sy, r_sz : in/out — src components, overwritten with normalized
* r_sq_y, r_sq_z : scratch — MAC2, MAC3 → die after stage 2 accumulate (r_sq_y may alias r_lzcr post-stage-2)
* r_recip_est : ≡ r_sqmag — multi-purpose (|v|² → shift-input → sqrtbl entry)
* r_lzcr : LZCR value (alive across stage 3 srav path)
* r_shift : (31 - LZCR & ~1) / 2 — final srav amount (stages 3-4)
* r_tmp : scratch — shift count, branch target, lookup addr, table base
*
* GPR ccount peak: 9.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
* Words: ~35 (pending re-gen; matches libgte 0x800160a0..0x8001615c at +/- 0-2 words).
*
* Sqrtbl address: link-time constant `&gte_normalize_sqrtbl`, split via >>16 and &0xFFFF. */
// ─── Stage 1: mtc2 src → IR1/2/3, SQR fires (MAC1/2/3 = IR², IR ← MAC saturated) ───
// Componentized equivalent: mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z).
// We inline for GPR-pressure reasons (see file-level comment).
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop, gte_cmdw_sqr,
// ─── Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS ───
// Note: r_recip_est first used as the sum accumulator (= |v|²), which is also what LZCS needs.
gte_mv_from_data_r(r_sq_y, C2_MAC1), /* r_sq_y = MAC1 = sx² */
gte_mv_from_data_r(r_sq_z, C2_MAC2), /* r_sq_z = MAC2 = sy² */
gte_mv_from_data_r(r_recip_est, C2_MAC3), /* r_recip_est = MAC3 = sz² */
nop, /* MFC2→GPR load delay (1 slot) */
add_u(r_recip_est, r_recip_est, r_sq_z), /* r_recip_est += sy² */
add_u(r_recip_est, r_recip_est, r_sq_y), /* r_recip_est += sx² (sum = |v|²) */
gte_mv_to_data_r( r_recip_est, C2_LZCS), /* LZCS = |v|² */
nop2,
gte_mv_from_data_r(r_lzcr, C2_LZCR), /* r_lzcr = LZCR (count of leading bits) */
nop, /* MFC2→GPR load delay (1 slot) */
// ─── Stage 3: compute shift amount, align |v|² to bit 24, lookup 1/|v| ───
// Matches libgte `bltz +0x10 ; nop ; b +0x14 ; sllv t4,v0,t3` pattern:
// - bltz TAKEN → nop (BD), jump to srav_path; sllv SKIPPED
// - bltz !TAKEN → nop (BD), b +0x14 jumps to aligned_done; sllv (BD of b) executes
and_i( r_lzcr, r_lzcr, -2), /* r_lzcr &= ~1 (force even for halving) */
li_s( r_shift, 31), /* r_shift = 31 */
sub_s( r_shift, r_shift, r_lzcr), /* r_shift = 31 - LZCR */
shift_aright( r_shift, r_shift, 1), /* r_shift = (31 - LZCR) / 2 */
add_si( r_tmp, r_lzcr, -24), /* r_tmp = LZCR - 24 (signed, for branch) */
branch_lt_zero(r_tmp, atom_offset(srav_path, aligned_done)), nop,
jump_rel( atom_offset(aligned_done, srav_path)),
shift_lleft_var(r_recip_est, r_recip_est, r_tmp), /* BD-slot of branch_equal: r_recip_est = |v|² << (LZCR - 24) */
atom_label(srav_path) /* SRAV path: |v|² is small (top bit < bit 24) */
li_s( r_tmp, 24),
sub_s( r_tmp, r_tmp, r_lzcr), /* r_tmp = 24 - LZCR */
shift_aright_var(r_recip_est, r_recip_est, r_tmp), /* r_recip_est = |v|² >> (24 - LZCR) */
atom_label(aligned_done) /* Both paths converge here with |v|² aligned to bit 24 */
/* r_recip_est now holds |v|² aligned to bit 24 — convert to byte offset, -64 to skip zero pad. */
add_si( r_recip_est, r_recip_est, -64),
shift_lleft( r_recip_est, r_recip_est, 1), /* r_recip_est *= 2 (half-word index) */
/* Reference OUR local sqrtbl via &-address split. Compiler/linker resolves both halves. */
load_upper_i( r_tmp, u4_hi(& gte_normalize_sqr_tbl)), /* lui */
or_i_self( r_tmp, u4_lo(& gte_normalize_sqr_tbl)), /* ori */
add_u( r_tmp, r_tmp, r_recip_est), /* r_tmp = sqrtbl base + byte offset (matches libgte 0x80016118: addu t5,t5,t4) */
load_half( r_recip_est, r_tmp, 0), /* r_recip_est = sqrtbl[r_recip_est] = 1/|v| estimate */
nop, /* retire load_half before MTC2 (matches libgte 0x80016120: nop) */
// ─── Stage 4: mtc2 IR0..3, GPF (MAC = IR0*IR), mfc2 MAC, srav finalize ───
// Componentized equivalent: mac_gte_gpf_scale.
gte_mv_to_data_r(r_recip_est, C2_IR0), /* IR0 = 1/|v| estimate */
gte_mv_to_data_r(r_sx, C2_IR1), /* IR1 = src.x */
gte_mv_to_data_r(r_sy, C2_IR2), /* IR2 = src.y */
gte_mv_to_data_r(r_sz, C2_IR3), /* IR3 = src.z */
nop2, /* COP2 transfer latency (2 slots) */
gte_cmdw_gpf,
gte_mv_from_data_r(r_sx, C2_MAC1), /* MAC1 → r_sx (overwrites src.x with raw reciprocal-scaled) */
gte_mv_from_data_r(r_sy, C2_MAC2),
gte_mv_from_data_r(r_sz, C2_MAC3),
shift_aright_var(r_sx, r_sx, r_shift),
shift_aright_var(r_sy, r_sy, r_shift),
shift_aright_var(r_sz, r_sz, r_shift),
})
#pragma endregion MACs (Mips Atom Components) #pragma endregion MACs (Mips Atom Components)
#pragma region Bsked Atoms #pragma region Bsked Atoms
typedef Struct_(Binds_SetGteWorld) { typedef Struct_(Binds_SetGteMT3S2S4) {
M3_S2* transform; MT3_S2S4* transform;
}; };
internal MipsAtom_(set_gte_world) atom_info( internal MipsAtom_(set_gte_mt3s2s4) atom_info(
atom_bind(Binds_SetGteWorld) atom_bind(Binds_SetGteMT3S2S4)
, atom_reads(R_TapePtr) , atom_reads(R_TapePtr)
){ ){
/* Pop matrix address from tape into R_T3 ($11) */ /* Pop matrix address from tape into R_T3 ($11) */
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)), load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)), add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */ /* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4), load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12), gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
+54 -13
View File
@@ -161,6 +161,8 @@ enum {
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */ gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
gte_cmd_op = 0x0C, /* Outer Product */ gte_cmd_op = 0x0C, /* Outer Product */
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */ gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
/* --- GTE Command Bit-Field Layout --- /* --- GTE Command Bit-Field Layout ---
* A GTE command word (sent to COP2 with RS=1) is laid out as: * A GTE command word (sent to COP2 with RS=1) is laid out as:
@@ -171,8 +173,7 @@ enum {
* +------------+--+-----+------+------+------+------+---+--------+----------+ * +------------+--+-----+------+------+------+------+---+--------+----------+
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/ * \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
* *
* Shifts/masks below are the *bit positions* and *bit widths* of each * Shifts/masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
* configurable field, used by the ENC_GTE_CMD encoder.
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h. * Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
*/ */
@@ -182,6 +183,12 @@ enum {
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3, gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1, gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F, gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
gte_shift_fake_cmd = 20,
gte_width_fake_cmd = 5,
gte_mask_fake_cmd = 0x1F,
}; };
/* --- GTE Control Register Indices (for ctc2/cfc2) --- /* --- GTE Control Register Indices (for ctc2/cfc2) ---
@@ -243,10 +250,10 @@ enum { _C2_OPS_ = 0
* bit 1 (0x02): register class — 0 = data, 1 = control * bit 1 (0x02): register class — 0 = data, 1 = control
* bit 2 (0x04): direction — 0 = read, 1 = write * bit 2 (0x04): direction — 0 = read, 1 = write
* *
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h * The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
* (which target the data register file on any coprocessor). * (which target the data register file on any coprocessor).
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2) * They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
* and so the encoding lives next to its only consumer (this header). * and so the encoding is next to its only consumer (this header).
* *
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */ * Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
enum { _C2_TX_SUBS_ = 0 enum { _C2_TX_SUBS_ = 0
@@ -309,13 +316,13 @@ enum { _C2_TX_SUBS_ = 0
/* GTE Command Format /* GTE Command Format
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO). * Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
* The lower 25 bits are the GTE-specific command payload. * Lower 25 bits are GTE-specific command payload.
* *
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h: * The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece * Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
* (handy for state-driven MVMVA emitters that vary one field at a time). * (handy for state-driven MVMVA emitters that vary one field at a time).
* *
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go. * `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
* It just ORs the per-field encoders together. */ * It just ORs the per-field encoders together. */
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25)) #define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
@@ -325,7 +332,8 @@ enum { _C2_TX_SUBS_ = 0
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v ) #define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv ) #define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm ) #define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd) #define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd ) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) (((x) & gte_mask_fake_cmd) << gte_shift_fake_cmd)
/* Composite: all six GTE fields + the COP2/CO base. */ /* Composite: all six GTE fields + the COP2/CO base. */
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \ #define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
@@ -363,11 +371,11 @@ enum { _C2_TX_SUBS_ = 0
* (the perspective divide happens regardless of `sf`). * (the perspective divide happens regardless of `sf`).
* *
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear), * If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops * PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided), * The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
* `nclip` ends up wrong, and the triangle is culled. * `nclip` ends up wrong, and the triangle is culled.
* *
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years. * So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source. * NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
* -------------------------------------------------------------------------- * --------------------------------------------------------------------------
*/ */
@@ -378,11 +386,45 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip)) #define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op )) #define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */ #define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */ #define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva)) #define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
/* SQR / GPF cosmetic-bits compat helpers.
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
* The hardware ignores these bits (per PSX-SPX line 48). */
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
/* SQR — Square Vector.
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
* bit 19 sf=0
* bit 10 lm=1
* bits 5-0 cmd=0x28=SQR
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
/* GPF — General-purpose Interpolation.
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf*12)
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
* bit 19 sf=0
* bit 10 lm=0
* bits 5-0 cmd=0x3D=GPF
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps #define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt #define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */ /* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20) #define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
@@ -433,7 +475,6 @@ enum {
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset) #define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders /* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
*
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in * Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern). * (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
* *
+96 -66
View File
@@ -12,68 +12,57 @@
#endif #endif
#pragma region Tape Drive #pragma region Tape Drive
/* ----------------------------------------------------------------------------- /* -----------------------------------------------------------------------------------------------------------
* TAPE DRIVE ABI * TAPE DRIVE ABI
* ----------------------------------------------------------------------------- * -----------------------------------------------------------------------------------------------------------
* Note(Ed): One of the main purposes of this codebase is to help me * Note(Ed): One of the main purposes of this codebase is to help me learn this,
* learn this, as such the information below may be entirely realized * as such the information below may not* be entirely realized or finalized conceptually.
* or finalized conceptually. * -----------------------------------------------------------------------------------------------------------
* ----------------------------------------------------------------------------- * This ABI and its associated legos were directly inspired by researching the work of
* This ABI and its associated legos were directly inspired by researching * Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
* the work of Timothy Lottes and Onat Türkçüoğlu; along with many others. * directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
* It's the simplest bootstrap of a a directly executed chain of assemby * These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
* arrays (Atoms) that terminate with a yield sequence to the next atom.
* These eventually lead to a terminal atom for the tape which is defined
* below as "tape_exit".
* *
* This behaves as one of the simplest runtime harnesses ontop of a * This behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
* host-enviornment's execution engine to author and compose programs with. * to author and compose programs with. From here various conventions can be further applied.
* From here various conventions can be further applied. * To make things easier to understand it may be better to focus on what this ABI does not have.
* To make things easier to understand it may be better to focus on what this * It does not have have any branching within the tape but relative branches within atoms or between atoms.
* ABI does not have. It does not have have any branching within the tape but * Branching nearly is always downstream. Stack usage is non-existent.
* relative branches between atoms. Branching nearly is always downstream. * Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
* Stack usage is non-existent. Push/Pop, FIFO, or Arena/Bump data structures * In it's current form with the C11 macro dsl, the user also has fullfill manual register allocation per atom.
* are used by atoms explicitly. In it's current form withe C11 macro dsl,
* the user also has to do manual register allocation per atom.
* *
* One of the remarkable things about utilizing this abi is its essentially * One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
* interopable with CPUs, GPUs, FPGA, or, basically anything * or, basically anything from the 5th generation consoles and onward.
* from the 5th generation consoles and onward. * The ABI directly reflects how all computational hardware must be architected in order to execute
* The ABI directly reflects how all computational hardware must be architected * digital logic effectively on current era tech.
* in order to execute digital logic effectively on current era tech. * On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
* On the PS1 we don't have access to a few features like multi-threading, * but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
* speculative execution, or L3 cache; but, we can set the foundation for legoing * and core atoms to take those newer hardware features into account. For example, you can easily expand
* whats required baseline wise for eventually expanding the harness and core atoms * this to support wave-based execution model on a PS2 or PS3. Not having a stack or
* to take those newer hardware features into account. For example, you can easily * automatic register allocation means the user cannott ignore excessive argument shuffle across workload or
* expand this to support wave-based execution model on a PS2 or PS3. * waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
* Not having a stack or automatic register allocation means the user can't ignore
* excessive argument shuffle across workload or waves and thier phases.
* Crossing ABI boundaries to other runtimes that do has an obviouss penalties.
* *
* Learning data-oreinted code becomes a natural progression. Your not fighting * Learning data-oreinted code becomes a natural progression. Your not fighting a stack-based procedural
* a stack-based procedural paradigm that wants to argument shuffle on the stack * paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
* by lack of constraints on how the user may "call" a procedure. The user doesn't * on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
* have to hammer down "rules" or patterns to know how to massage the compiler * hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
* to get the asesmbly into its natural form. The form is obvious, and once * the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
* the user gets to author their compoonents it becomes a game of tetris. * it becomes a game of tetris.
* *
* Another feature is this ABI is very compatible with bootstrapping and developing * Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
* simple toolchains built off of bit-packed annotated command streams the user can * of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
* directly author, maintatain, and immediately execute. That being a color forth. * That being like a color forth, or maybe something more familar like an immediate mode library
* This can make the tetris less of a chore with some helpful policy generation for * for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
* allocation of registers, helping to choose resuable components, designing DSL on * generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
* the fly, etc. * -----------------------------------------------------------------------------------------------------------
* ----------------------------------------------------------------------------- * TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
* TODO(Ed): We ned pretty ascii diagrams and proper guides, articles, etc. * -----------------------------------------------------------------------------------------------------------
* ----------------------------------------------------------------------------- * For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
* For now this thing is just functioning and I'm abusing C11 + a lua metaprogram * a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
* to help establish a hybrid toolchain to ideate on a traditional text-based * If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
* authoring UX for this paradigm. * (just copying ram to filesystem), I can author a color forth to mess around with.
* If pcsx-redux gets me viable hot-reload and persistent data storage beyond * With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
* save-states (just copying ram to filesystem). I can author a color forth to * but I think this codebase most likely has a pretty ergonomic flavor worst case...
* mess around with, with an editor in-emulator or on the actual machine itself.
* Assembly is tedius, but I think this codebase most likely has some of the most,
* ergonomic you can come across..
* */ * */
/* Register Allocation Info */ /* Register Allocation Info */
enum { enum {
@@ -117,6 +106,12 @@ typedef Slice_(MipsCode);
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield. typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
#define MipsAtom_(sym) MipsCode sym [] align_(4) = #define MipsAtom_(sym) MipsCode sym [] align_(4) =
// Used for atoms with value-args
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
// expands to:
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; }
#define MipsAtom_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); }
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names). // Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
// MipsAtomComp_(ac_X) { body } // MipsAtomComp_(ac_X) { body }
// expands to: // expands to:
@@ -129,8 +124,14 @@ typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); } // FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); } #define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the // Used for components with value-args (e.g., ac_format_f3_color).
file contains line-numbered content. Files containing only: // FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
// expands to:
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
// #define MipsAtomComp_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); }
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
Files containing only:
- `MipsAtomComp_` static-array declarations, or - `MipsAtomComp_` static-array declarations, or
- `MipsAtomComp_Proc_` (force-inline) function bodies whose line info gets - `MipsAtomComp_Proc_` (force-inline) function bodies whose line info gets
attributed to the call site at the include point are otherwise omitted from the file table, attributed to the call site at the include point are otherwise omitted from the file table,
@@ -192,11 +193,13 @@ FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; } FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; } FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; } FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; } FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
#define tb_emit_(atom) tb_emit(& tb, atom) #define tb_emit_(atom) tb_emit(& tb, atom)
#define tb_data_(field, data) tb_data(& tb, u4_(data)) #define tb_data_(field, data) tb_data(& tb, u4_(data))
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), tb->used); tb->used += atoms.len; }
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; } FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; } FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit)) #define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
@@ -244,22 +247,49 @@ typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 u
// Whatever the builder is writting to should most likely coresspond // Whatever the builder is writting to should most likely coresspond
// to something that can fit within instruction cache? // to something that can fit within instruction cache?
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) { FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode code) {
assert(ab->capacity - ab->used - code->len); assert(ab->capacity - ab->used - code.len);
mem_copy(ab->start, u4_(code->ptr), code->len); mem_copy(ab->start, u4_(code.ptr), code.len);
mem_bump(ab->start, ab->capacity, & ab->used, code->len); mem_bump(ab->start, ab->capacity, & ab->used, code.len);
} }
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac)) #define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
// When done authoring, utilize this to cap-off the atom // When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
FI_ void atombuilder_end(MipsAtomBuilder_R ab) { FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield)); mem_copy(ab->start, u4_(ac_yield), S_(ac_yield));
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield)); mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
} }
#define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used} #define mipsatom_from_builder(ab) C_(MipsAtom*, (ab).start)
#pragma endregion Mips Atom Builder #pragma endregion Mips Atom Builder
#pragma region Mips Atom Procs
#if 0
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
FI_ void sync_prim_arean_proc_demo(MipsAtomBuilder_R ab, U4 r_extra, U4 add_amnt_extra)
MipsAtom_Proc_(sync_primitive_arena_proc_demo, ab, atom_info(atom_bind(Binds_SyncPrimitiveArena)
, atom_reads( R_TapePtr, R_PrimCursor)
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
add_ui_self(r_extra, add_amnt_extra), // extra op for demonstration purposes.
mac_yield()
})
void demo_make_make_and_emit_atom(TapeBuilder* tb, MipsAtomBuilder* ab){
sync_prim_arean_proc_demo(ab, R_T4, 4);
tb_emit(tb, mipsatom_from_builder(ab[0]));
}
#endif
#pragma endregion Mips Atom Procs
#pragma region Baked Mips Atoms #pragma region Baked Mips Atoms
// These atoms are resolved at compile time and are (usually) statically linked readonly data. // These atoms are resolved at compile time and are (usually) statically linked readonly data.
+18
View File
@@ -19,6 +19,24 @@ FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_
store_half(rt_y, base, offset + O_(V2_S2,y)), store_half(rt_y, base, offset + O_(V2_S2,y)),
}) })
FI_ Slice_MipsCode ac_load_v3s4(U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, {
load_word( rs_x, r_base, O_(V3_S4,x)),
load_word( rs_y, r_base, O_(V3_S4,y)),
load_word( rs_z, r_base, O_(V3_S4,z)),
})
FI_ Slice_MipsCode ac_store_v3s4(U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, {
store_word(rt_x, base, offset + O_(V3_S4,x)),
store_word(rt_y, base, offset + O_(V3_S4,y)),
store_word(rt_z, base, offset + O_(V3_S4,z)),
})
FI_ Slice_MipsCode ac_sub_v3s4(U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, {
sub_s(rds_x, rds_x, rt_x),
sub_s(rds_y, rds_y, rt_y),
sub_s(rds_z, rds_z, rt_z),
})
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, { FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
store_half(rt_x, base, offset + O_(Rect_S2,x)), store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)), store_half(rt_y, base, offset + O_(Rect_S2,y)),
+55 -5
View File
@@ -7,6 +7,18 @@
#define max(A, B) (((A) > (B)) ? (A) : (B)) #define max(A, B) (((A) > (B)) ? (A) : (B))
#define clamp_bot(X, B) max(X, B) #define clamp_bot(X, B) max(X, B)
/* Convention
<Type> ## <Width> _ <Component Type> ## <Component Width>
For types with compound data (Ex: Rotation Matrix & Translation):
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
A: Array
V: Vector
R: Range
M: Matrix
T: Translation
*/
enum { enum {
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset. v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
}; };
@@ -26,23 +38,38 @@ typedef Struct_(Extent2_S4) { S4 width; S4 height; };
typedef Struct_(V2_U1) { U1 x; U1 y; }; typedef Struct_(V2_U1) { U1 x; U1 y; };
typedef Struct_(V2_S2) { S2 x; S2 y; }; typedef Struct_(V2_S2) { S2 x; S2 y; };
typedef Struct_(V2_S4) { S4 x; S4 y; }; typedef Struct_(V2_S4) { S4 x; S4 y; };
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; }; typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; }; typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; typedef V3_S4 P3_S4;
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; }; typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; }; typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; }; typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
/* RGA(Lengyel) reserved names (deferred):
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
typedef Array_(V2_U1, 2);
typedef Array_(V2_S2, 2); typedef Array_(V2_S2, 2);
typedef Array_(V2_S2, 3); typedef Array_(V2_S2, 3);
typedef Array_(V2_S2, 4); typedef Array_(V2_S2, 4);
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
#define v2s2(x,y) (V2_S2){x,y} #define v2s2(x,y) (V2_S2){x,y}
#define v3s2(x,y,z) (V3_S2){x,y,z,0} #define v3s2(x,y,z) (V3_S2){x,y,z,0}
#define v3s4(x,y,z) (V3_S4){x,y,z,0} #define v3s4(x,y,z) (V3_S4){x,y,z,0}
@@ -61,5 +88,28 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[2] += b[2] >> 1; (out_a[0])[2] += b[2] >> 1;
} }
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] -= b[0];
(out_a[0])[1] -= b[1];
(out_a[0])[2] -= b[2];
}
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] -= b[0] >> 1;
(out_a[0])[1] -= b[1] >> 1;
(out_a[0])[2] -= b[2] >> 1;
}
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
(out_a[0])[0] *= b[0];
(out_a[0])[1] *= b[1];
(out_a[0])[2] *= b[2];
}
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); } FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (pcast(A3_S4_R, out_a), pcast(A3_S4, b)); }
+1 -5
View File
@@ -1,6 +1,7 @@
#ifdef INTELLISENSE_DIRECTIVES #ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h" # include "gen/macs.h"
# include "gen/offsets.h" # include "gen/offsets.h"
# include "bios.h"
# include "lottes_tape.h" # include "lottes_tape.h"
#endif #endif
@@ -8,11 +9,6 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
#pragma region Baked Atoms #pragma region Baked Atoms
enum {
bios_flushcache = 0x44,
bios_table_addr = 0xA0,
};
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0). /* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack): * Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
* 1. sp -= 8; sw $ra, 4($sp) ; save RA * 1. sp -= 8; sw $ra, 4($sp) ; save RA
+12 -11
View File
@@ -348,6 +348,12 @@ enum { _BitOffsets = 0
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl) #define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra) #define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
/* Shift Variable — register-shift forms.
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll) #define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt) #define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
@@ -366,20 +372,18 @@ enum { _BitOffsets = 0
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline. * WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset. * The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve. * The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
*
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s * `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR * - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field. * - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
*/ */
#define jump(off) enc_i(op_j, R_0, R_0, (off)) #define jump(off) enc_i(op_j, R_0, R_0, (off))
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`). /* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. * MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
*/
#define jump_rel(off) branch_equal(R_0, R_0, (off)) #define jump_rel(off) branch_equal(R_0, R_0, (off))
/* call_addr off — jump-and-link to immediate address. /* call_addr off — jump-and-link to immediate address.
*
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target. * Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom. * For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register. * Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
@@ -397,13 +401,7 @@ enum { _BitOffsets = 0
* sub_s / sub_u → sub / subu * sub_s / sub_u → sub / subu
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO) * mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
* div_s / div_u → div / divu (LO = quot, HI = rem) * div_s / div_u → div / divu (LO = quot, HI = rem)
* */
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
#undef add_s
#undef sub_s
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add) #define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu) #define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub) #define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
@@ -458,6 +456,9 @@ enum { _BitOffsets = 0
#define nop shift_lleft(rdiscard, rdiscard, 0) #define nop shift_lleft(rdiscard, rdiscard, 0)
#define nop2 nop, nop #define nop2 nop, nop
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm)) #define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm)) #define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
+84 -73
View File
@@ -9,6 +9,34 @@
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c); ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
#pragma region MACs (Mips Atom Components)
FI_ Slice_MipsCode ac_pad_set_centered_axes(U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, {
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF),
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF),
store_word( r_scratch, r_state, O_(PadState,axes)),
})
FI_ Slice_MipsCode ac_pad_set_id_byte(U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, {
add_ui( r_id, R_0, id_value),
store_byte(r_id, r_state, O_(PadState,id)),
})
FI_ Slice_MipsCode ac_pad_set_status(U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, {
add_ui( r_tmp, R_0, pad_status),
store_word(r_tmp, r_state, O_(PadState,status)),
})
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, {
nor_u( r_buttons, r_buttons, R_0),
store_half( r_buttons, r_pad_state, O_(PadState, buttons)),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms #pragma region Baked Atoms
/* ----- pad_bios_snapshot ----- /* ----- pad_bios_snapshot -----
@@ -35,7 +63,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
*/ */
enum { enum {
R_PadRaw = R_T0 atom_reg atom_type(U1), R_PadRaw = R_T0 atom_reg atom_type(U1),
R_PadState = R_T1 atom_reg, R_PadState = R_T1 atom_reg atom_type(PadState*),
R_RawStatus = R_T2 atom_reg, R_RawStatus = R_T2 atom_reg,
R_RawId = R_T3 atom_reg, R_RawId = R_T3 atom_reg,
}; };
@@ -44,8 +72,8 @@ typedef Struct_(Binds_PadBiosSnapshot) {
PadState* state; PadState* state;
}; };
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot) internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr) , atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId, R_T4, R_T5, R_TapePtr) , atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
) { ) {
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */ /* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)), load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
@@ -53,111 +81,97 @@ internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)), add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
/* === Read raw[0] (status) + raw[1] (id) */ /* === Read raw[0] (status) + raw[1] (id) */
load_byte_u(R_RawStatus, R_PadRaw, 0), load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
load_byte_u(R_RawId, R_PadRaw, 1), load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */ atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
add_ui(R_T4, R_0, 0xFF), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)), add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes. /* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */ * If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
atom_label(disconnected) /* === Disconnected body. */ atom_label(disconnected) /* === Disconnected body. */
/* R_T4 = PadStatus_Disconnected from snap_root BD-slot. */ mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
store_word(R_T4, R_PadState, O_(PadState,status)), store_half( R_0, R_PadState, O_(PadState,buttons)),
store_half(R_0, R_PadState, O_(PadState,buttons)), mac_pad_set_centered_axes(R_PadState, R_T4),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */ mac_pad_set_id_byte( R_PadState, R_RawId, PadRawStatus_Timeout),
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(disconnected, snap_end)), jump_rel(atom_offset(disconnected, snap_end)),
/* BD-slot: load next atom's entry point (replaces the nop). /* BD-slot: load next atom's entry point (replaces the nop).
* The unconditional branch always jumps to snap_end, where mac_yield_tail() * Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
* transfers control to R_AtomJmp without re-loading it. */
mac_yield_load(), mac_yield_load(),
atom_label(skip_disconnected) atom_label(skip_disconnected)
/* === Case 2: Pending (status == 0 && id == 0) /* === Case 2: Pending (status == 0 && id == 0)
* Combined check: if (status | id) != 0 then skip to id_dispatch. * Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
* Falls through to the Pending case only when both are zero. */
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)), or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes. /* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui harmless. */ * If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
atom_label(pending) /* === Pending body */ atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
/* R_T4 = PadStatus_Pending from case_2 BD-slot. */ mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
store_word(R_T4, R_PadState, O_(PadState,status)), store_half( R_0, R_PadState, O_(PadState,buttons)),
store_half(R_0, R_PadState, O_(PadState,buttons)), mac_pad_set_centered_axes(R_PadState, R_T4),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */ store_byte(R_RawId, R_PadState, O_(PadState,id)),
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
store_byte( R_RawId, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(pending, snap_end)), jump_rel(atom_offset(pending, snap_end)),
mac_yield_load(), mac_yield_load(),
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */ atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
add_ui(R_T4, R_0, 0x41), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)), add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes. /* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */ * If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
/* === Digital body (status, buttons normalize, axes=0x80, id, branch. */ /* === Digital body (status, buttons normalize, axes=0x80, id, branch.
/* R_T4 = PadStatus_Digital from id_dispatch BD-slot. */ * R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
store_word( R_T4, R_PadState, O_(PadState,status)), * R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
/* Fill R_T4's load-delay slot with the 0x80808080 axes constant into R_T5 load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
* (R_T5 is dead on this path; it's only consumed at the analog_pad range check). */ load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo), /* fills the buttons-load's delay slot (doesn't read R_T4) */
load_upper_i(R_T5, 0x8080), or_i_self(R_T5, 0x8080), mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
nor_u( R_T4, R_T4, R_0), /* raw_buttons is already in host bit order; no swap needed */ store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
store_half( R_T4, R_PadState, O_(PadState,buttons)), mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */
store_word( R_T5, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0x41),
store_byte( R_T4, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(id_dispatch, snap_end)), jump_rel(atom_offset(id_dispatch, snap_end)),
mac_yield_load(), mac_yield_load(),
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/ atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
add_ui(R_T4, R_0, 0x53), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)), add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes. /* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */ * If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
atom_label(analog_stick) /* === AnalogStick body atom_label(analog_stick) /* === AnalogStick body
* Axes are loaded as two halfwords: raw[6..7] → left_xy (sh at offset 8), raw[4..5] → right_xy (sh at offset 10). * R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
* R_T5 holds left_xy / id-value in turn (it's dead on this path — only consumed at the analog_pad range check). */ * R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
/* R_T4 = PadStatus_AnalogStick from try_analog_stick BD-slot. */ * R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
store_word( R_T4, R_PadState, O_(PadState,status)), * The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
load_half_u( R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */ mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
load_half_u( R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot (doesn't read R_T4) */ load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */ load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
store_half( R_T4, R_PadState, O_(PadState,buttons)), mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
load_half_u( R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */ load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */ store_half( R_T5, R_PadState, O_(PadState, left)),
store_half( R_T4, R_PadState, O_(PadState,right_x)), store_half( R_T4, R_PadState, O_(PadState, right)),
add_ui( R_T5, R_0, 0x53), /* R_T5 = id value (clobbers left_xy, already stored) */ mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
store_byte( R_T5, R_PadState, O_(PadState,id)),
jump_rel(atom_offset(analog_stick, snap_end)), jump_rel(atom_offset(analog_stick, snap_end)),
mac_yield_load(), mac_yield_load(),
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */ atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
and_i( R_T4, R_RawId, 0xF0), and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
add_ui( R_T5, R_0, 0x70), add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)), branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes. /* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */ * If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
atom_label(analog_pad) /* === AnalogPad body atom_label(analog_pad) /* === AnalogPad body
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path). */ * Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
/* R_T4 = PadStatus_AnalogPad from try_analog_pad BD-slot. */ * The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
store_word( R_T4, R_PadState, O_(PadState,status)), * Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
load_half_u(R_T4, R_PadRaw, 2 * S_(U1)), /* R_T4 = raw_buttons */ mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
load_half_u(R_T5, R_PadRaw, 6 * S_(U1)), /* R_T5 = left_xy; fills R_T4's load-delay slot */ load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
nor_u( R_T4, R_T4, R_0), /* R_T4 = ~raw_buttons */ load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
store_half( R_T4, R_PadState, O_(PadState,buttons)), mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
load_half_u(R_T4, R_PadRaw, 4 * S_(U1)), /* R_T4 = right_xy; fills R_T5's load-delay slot */ load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
store_half( R_T5, R_PadState, O_(PadState,left_x)), /* R_T5 settled, store left_xy */ store_half( R_T5, R_PadState, O_(PadState, left)),
store_half( R_T4, R_PadState, O_(PadState,right_x)), store_half( R_T4, R_PadState, O_(PadState, right)),
store_byte( R_RawId, R_PadState, O_(PadState,id)), store_byte( R_RawId, R_PadState, O_(PadState, id)),
jump_rel(atom_offset(analog_pad, snap_end)), jump_rel(atom_offset(analog_pad, snap_end)),
mac_yield_load(), mac_yield_load(),
@@ -166,11 +180,8 @@ atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the
add_ui( R_T4, R_0, PadStatus_Unsupported), add_ui( R_T4, R_0, PadStatus_Unsupported),
store_word(R_T4, R_PadState, O_(PadState,status)), store_word(R_T4, R_PadState, O_(PadState,status)),
store_half(R_0, R_PadState, O_(PadState,buttons)), store_half(R_0, R_PadState, O_(PadState,buttons)),
/* axes = 0x80808080 (centered) — single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y). */ mac_pad_set_centered_axes(R_PadState, R_T4),
load_upper_i(R_T4, 0x8080), or_i_self(R_T4, 0x8080), mac_pad_set_id_byte( R_PadState, R_RawId, PadUnknownId_Sentinel),
store_word( R_T4, R_PadState, O_(PadState,left_x)),
add_ui( R_T4, R_0, 0xFF), /* 0xFF sentinel: "unknown id" */
store_byte( R_T4, R_PadState, O_(PadState,id)),
/* Fall through to snap_end. */ /* Fall through to snap_end. */
atom_label(no_jump_fallthrough) atom_label(no_jump_fallthrough)
+78
View File
@@ -0,0 +1,78 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "dsl.h"
# include "gcc_asm.h"
# include "mips.h"
# include "bios.h"
# include "pad.h"
#endif
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
*
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
* The C-level writes after the call re-load the pointers from their callee-saved homes.
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
// Use enums.
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
* $a0 = raw0 (rgcc-bound; survives the sequence below)
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, bios_pad_buffer_size), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, bios_pad_buffer_size), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, bios_init_pad_2), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
u1_v(raw0)[0] = 0xFF;
u1_v(raw1)[0] = 0xFF;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
asm volatile(
asm_words(
add_ui( rtmp_1, rdiscard, bios_start_pad_2), /* $t1 = 0x13 */
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
}
+63 -22
View File
@@ -5,8 +5,9 @@
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421. /* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
* Wire is active-low (0 = pressed). * Wire is active-low (0 = pressed).
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */ * The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
enum { * active-low-to-active-high inversion is applied bit-by-bit. */
typedef Enum_(U2, PadBtns) {
Bit_(Pad_Select, 0), Bit_(Pad_Select, 0),
Bit_(Pad_L3, 1), Bit_(Pad_L3, 1),
Bit_(Pad_R3, 2), Bit_(Pad_R3, 2),
@@ -32,18 +33,22 @@ enum {
Pad1 = 1 << PadId_Offset, Pad1 = 1 << PadId_Offset,
}; };
#define pad0_(btn_id) (btn_id << Pad0) /* =============================================================================
#define pad1_(btn_id) (btn_id << Pad1)
/* ============================================================
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h)) * BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
* ============================================================ */ * ============================================================================= */
enum { enum {
PAD_BIOS_RAW_SIZE = 0x22, PAD_BIOS_RAW_SIZE = 0x22,
}; };
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
typedef Struct_(PadBiosRaw) { typedef Struct_(PadBiosRaw) {
U1 bytes[PAD_BIOS_RAW_SIZE]; U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
V2_U1 right; /* offset 4-5 (right stick x, y) */
V2_U1 left; /* offset 6-7 (left stick x, y) */
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
}; };
typedef Enum_(U4, PadStatus) { typedef Enum_(U4, PadStatus) {
@@ -56,18 +61,54 @@ typedef Enum_(U4, PadStatus) {
PadStatus_Invalid, PadStatus_Invalid,
}; };
/* PadState — per-port normalized runtime state. /* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y) * PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction. * when the controller id does not match any known controller type.
* The struct size stays 12 bytes (unchanged from the prior order, * PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */ * four byte axes at PadState.left_x through PadState.right_y. */
typedef Struct_(PadState) { typedef Enum_(U1, PadRawStatus) {
PadStatus status; /* offset 0, size 4 (U4) */ PadRawStatus_Ok = 0x00,
U2 buttons; /* offset 4, size 2 */ PadRawStatus_Timeout = 0xFF,
U1 id; /* offset 6, size 1 */
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
U1 left_y; /* offset 9, size 1 */
U1 right_x; /* offset 10, size 1 */
U1 right_y; /* offset 11, size 1 */
}; };
typedef Enum_(U1, PadRawId) {
PadRawId_Digital = 0x41,
PadRawId_AnalogStick = 0x53,
PadRawId_AnalogPadMask = 0xF0,
PadRawId_AnalogPadValue = 0x70,
};
typedef Enum_(U1, PadUnknownId) {
PadUnknownId_Sentinel = 0xFF,
};
typedef Enum_(U4, PadAxisCentered) {
PadAxis_Centered_Hi = 0x8080,
PadAxis_Centered_Lo = 0x8080,
PadAxis_Centered_Word = 0x80808080U,
};
typedef Enum_(U1, PadDeadZone) {
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
};
typedef Struct_(PadAxes) {
V2_U1 left; /* offset 8-9 */
V2_U1 right; /* offset 10-11 */
};
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
typedef Struct_(PadState) {
PadStatus status; /* offset 0, (U4) */
PadBtns buttons; /* offset 4, */
U1 id; /* offset 6, */
byte_pad(1); /* offset 7, explicit pad to align the axes block */
union {
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
struct {
V2_U1 left; /* offset 8-9 */
V2_U1 right; /* offset 10-11 */
};
};
};
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
+23 -5
View File
@@ -64,9 +64,9 @@ typedef Struct_(Tile) {
Linear Algebra Linear Algebra
*/ */
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix"); MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix"); MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix"); MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
// Rotation, Translation, Perspective // Rotation, Translation, Perspective
@@ -99,5 +99,23 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
); );
} }
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix"); void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix"); void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
+9
View File
@@ -54,6 +54,15 @@ WORD_COUNT(gte_sw, 1)
WORD_COUNT(gte_cmdw_rtpt, 1) WORD_COUNT(gte_cmdw_rtpt, 1)
WORD_COUNT(gte_cmdw_nclip, 1) WORD_COUNT(gte_cmdw_nclip, 1)
WORD_COUNT(gte_avg_sort_z3, 1) WORD_COUNT(gte_avg_sort_z3, 1)
WORD_COUNT(gte_cmdw_sqr, 1)
WORD_COUNT(gte_cmdw_gpf, 1)
WORD_COUNT(shift_lleft_var, 1)
WORD_COUNT(shift_aright_var, 1)
WORD_COUNT(li_s, 1)
WORD_COUNT(and_i, 1)
WORD_COUNT(add_si, 1)
WORD_COUNT(branch_lt_zero, 1)
WORD_COUNT(sub_s, 1)
WORD_COUNT(sub_u, 1) WORD_COUNT(sub_u, 1)
WORD_COUNT(nop2, 2) WORD_COUNT(nop2, 2)
+19 -1
View File
@@ -8,7 +8,7 @@
#pragma region hello_camera #pragma region hello_camera
// --- atom: pad_apply_input (60 words) --- // --- atom: pad_input_cube_rotation (60 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6 #define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6 #define _atom_offset_dpad_right_exit_dpad_right 6
@@ -26,6 +26,24 @@ enum {
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick, atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
}; };
// --- atom: pad_input_cam (40 words) ---
#define _atom_offset_left_x_exit_left_x 3
#define _atom_offset_right_x_exit_right_x 3
#define _atom_offset_up_y_exit_up_y 3
#define _atom_offset_down_y_exit_down_y 3
#define _atom_offset_cross_z_exit_cross_z 3
#define _atom_offset_circle_z_exit_circle_z 3
enum {
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
};
// --- atom: cube_g4_face (76 words) --- // --- atom: cube_g4_face (76 words) ---
#define _atom_offset_cull_cube_g4_face_exit 41 #define _atom_offset_cull_cube_g4_face_exit 41
+118 -37
View File
@@ -180,26 +180,6 @@ internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads
mac_yield(), mac_yield(),
}; };
/* ----- pad_apply_input -----
* Reads pad[0].buttons + pad[0].left_x;
* Applies the input-semantics deltas to cube_rot.y + floor_rot.y:
* - D-pad Left: cube_rot.y += 30, floor_rot.y += 5
* - D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5
* - Analog stick X (dead zone 0x70..0x90):
* cube delta = (0x80 - left_x) >> 2 (range approx -32..+32)
* floor delta = (0x80 - left_x) >> 5 (range approx -4..+4)
* - D-pad + analog deltas add when used together.
*
* Convention:
* pad_state = 0 means no buttons active.
* The fail-safe zero-button value flows through unchanged, so a disconnected/fresh pad produces no rotation.
* The branch_le_zero pattern below matches the existing pad_input_demo convention (atom body lines 248/257).
*
* Signed-delta trick:
* load_byte_u zero-extends left_x to 32 bits; sub_u from 0x80 wraps to a SIGNED two's-complement value in the negative range;
* shift_aright (sra) then correctly sign-extends the shift for both positive (left_x < 0x80) and negative (left_x > 0x80) cases.
* Digital pads publish left_x = 0x80 → delta = 0 → no rotation, so the analog step is naturally a no-op for digital controllers.
*/
typedef Struct_(Binds_PadApplyInput) { typedef Struct_(Binds_PadApplyInput) {
PadState* state; PadState* state;
V3_S2* cube_rot; V3_S2* cube_rot;
@@ -210,7 +190,7 @@ enum {
R_CubeRot = R_T1 atom_reg, R_CubeRot = R_T1 atom_reg,
R_FloorRot = R_T2 atom_reg, R_FloorRot = R_T2 atom_reg,
}; };
internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput) internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr) , atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
, atom_writes( R_CubeRot, R_FloorRot) , atom_writes( R_CubeRot, R_FloorRot)
) { ) {
@@ -225,7 +205,7 @@ internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
// Note(Ed): Potential op with delay slot? // Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */ /* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, pad0_(Pad_Left)), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */ load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)), load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30), add_si( R_T4, R_T4, 30),
@@ -235,7 +215,7 @@ internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
atom_label(exit_dpad_left) atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */ /* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, pad0_(Pad_Right)), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */ load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
load_half( R_T3, R_FloorRot, O_(V3_S2,y)), load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30), add_si( R_T4, R_T4, -30),
@@ -246,21 +226,21 @@ internal MipsAtom_(pad_apply_input) atom_info(atom_bind(Binds_PadApplyInput)
/* Analog left-stick X: dead zone 0x70..0x90. /* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */ * Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly). /* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */ * set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
add_ui(R_T4, R_0, 0x70), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)), add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
add_ui(R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_low_active */ add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
atom_label(dead_check_upper) atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */ /* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left_x)), /* reload */ load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
add_ui( R_T4, R_0, 0x90), add_ui( R_T4, R_0, PadDeadZone_HighBound),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */ /* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
add_ui( R_T4, R_0, 0x80), /* BD-slot: pre-load 0x80 for dead_high_active */ add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
jump_rel(atom_offset(dead_zone_skip, exit_stick)), jump_rel(atom_offset(dead_zone_skip, exit_stick)),
mac_yield_load(), mac_yield_load(),
@@ -273,8 +253,7 @@ atom_label(dead_low_active)
/* R_T4 = cube_delta */ /* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2), shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
nop,
add_u( R_T0, R_T0, R_T4), add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)), store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap; /* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
@@ -295,8 +274,7 @@ atom_label(dead_high_active)
/* delta = 0x80 - left_x (signed negative). */ /* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */ shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
nop,
add_u( R_T0, R_T0, R_T4), add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)), store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
@@ -314,6 +292,109 @@ atom_label(exit_stick)
mac_yield_tail(), mac_yield_tail(),
}; };
enum {
R_Cam = R_T4 atom_reg,
R_CamPadState = R_T5 atom_reg,
};
typedef Struct_(Binds_PadInputCam) {
PadState* state;
Camera* cam;
};
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
, atom_writes(R_Cam)
) {
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_left_x)
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_right_x)
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_up_y)
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_down_y)
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_cross_z)
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_circle_z)
mac_yield_tail(),
};
enum {
R_LookAt = R_T0 atom_reg atom_type(MT3_S2S4*),
R_CamEye = R_T1 atom_reg atom_type(P3_S4*),
R_CamTarget = R_T2 atom_reg atom_type(P3_S4*),
R_WorldUp = R_T3 atom_reg atom_type(V3_S4*),
R_LkAt_Fwdx = R_T4 atom_reg atom_type(V3_S4*),
R_LkAt_Fwdy = R_T5 atom_reg atom_type(V3_S4*),
R_LkAt_Fwdz = R_T6 atom_reg atom_type(V3_S4*),
R_Eye_x = R_T7 atom_reg atom_type(V3_S4*),
R_Eye_y = R_T8 atom_reg atom_type(V3_S4*),
R_Eye_z = R_V0 atom_reg atom_type(V3_S4*),
R_LkAt_Up = R_T5 atom_reg atom_type(V3_S4*),
R_LkAt_Right = R_T6 atom_reg atom_type(V3_S4*),
R_AxisX = R_T7 atom_reg atom_type(V3_S4*),
R_AxisY = R_T8 atom_reg atom_type(V3_S4*),
R_AxisZ = R_T7 atom_reg atom_type(V3_S4*),
};
typedef Struct_(Binds_ResolveLookAt) {
MT3_S2S4* look_at;
P3_S4* eye;
P3_S4* target;
V3_S4* up_in;
};
internal MipsAtom_(resolve_look_at) atom_info(atom_bind(Binds_ResolveLookAt)) {
load_word(R_LookAt, R_TapePtr, O_(Binds_ResolveLookAt,look_at)),
load_word(R_CamEye, R_TapePtr, O_(Binds_ResolveLookAt,eye)),
load_word(R_CamTarget, R_TapePtr, O_(Binds_ResolveLookAt,target)),
load_word(R_WorldUp, R_TapePtr, O_(Binds_ResolveLookAt,up_in)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAt)),
// load look_at and eye, then subtract (get direction), then normalize to unit vector.
mac_load_v3s4(R_LkAt_Fwdx, R_LkAt_Fwdy, R_LkAt_Fwdz, R_LookAt, 0),
mac_load_v3s4(R_Eye_x, R_Eye_y, R_Eye_z, R_CamEye, 0),
mac_sub_v3s4( R_LkAt_Fwdx, R_LkAt_Fwdy, R_LkAt_Fwdz,
R_Eye_x, R_Eye_y, R_Eye_z),
// ac_normalize_v3s4(9 args): in-place normalize direction → unit vector.
// Reg-aliasing across the 4 stages: R_T7 = r_sq_y → r_lzcr, R_T8 = r_sq_z → r_shift,
// R_V0 = r_recip_est (always), R_V1 = r_tmp. r_sx/r_sy/r_sz = R_LkAt_Fwdx/y/z (in-place).
// mac_normalize_v3s4(R_LkAt_Fwdx, R_LkAt_Fwdy, R_LkAt_Fwdz,
// R_T7, R_T8,
// R_V0,
// R_T7, R_T8, R_V1),
mac_yield(),
};
enum { enum {
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */ R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */ R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
@@ -344,7 +425,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
mac_yield() mac_yield()
}; };
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline // cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
internal internal
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4), MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase), atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
@@ -363,7 +444,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)), branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer). /* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded — * If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later, only on the body path. */ * harmless because the OT entry that points to this prim is created later. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)), store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase), shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
@@ -379,7 +460,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
set_lt_u( R_AT, R_T1, R_AT), set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop, branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor), mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
mac_format_g4_color(R_PrimCursor, mac_format_g4_color(R_PrimCursor,
/* c0 magenta */ 0xFF, 0x00, 0xFF, /* c0 magenta */ 0xFF, 0x00, 0xFF,
/* c1 yellow */ 0xFF, 0xFF, 0x00, /* c1 yellow */ 0xFF, 0xFF, 0x00,
@@ -439,7 +520,7 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
set_lt_u( R_AT, R_T1, R_AT), set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop, branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white) mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */ mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */ add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
// Note(Ed): No bounds checking, should be checked before atom runs. // Note(Ed): No bounds checking, should be checked before atom runs.
// end: branch(bounds_chk) // end: branch(bounds_chk)
+133 -83
View File
@@ -26,10 +26,12 @@
#include "duffle/dsl.atom.h" #include "duffle/dsl.atom.h"
#include "duffle/lottes_tape.h" #include "duffle/lottes_tape.h"
#include "duffle/bios.h"
#include "duffle/psyq.h" #include "duffle/psyq.h"
#pragma endregion Duffle Headers #pragma endregion Duffle Headers
#pragma region Duffle TUs #pragma region Duffle TUs
#include "duffle/pad.c"
#include "duffle/math.atom.c" #include "duffle/math.atom.c"
#include "duffle/mips.atom.c" #include "duffle/mips.atom.c"
#include "duffle/gte.atom.c" #include "duffle/gte.atom.c"
@@ -61,7 +63,10 @@ typedef Struct_(SMemory) {
U4 MemTape[MemTape_Len]; U4 MemTape[MemTape_Len];
M3_S2 tform_world; MT3_S2S4 tform_world;
MT3_S2S4 tform_view;
Camera cam;
Ent_Cube cube; Ent_Cube cube;
Ent_Floor floor; Ent_Floor floor;
@@ -74,6 +79,9 @@ typedef Struct_(SMemory) {
global SMemory smem; global SMemory smem;
extern SMemory smem; extern SMemory smem;
#define pad0_btn_(btn) btn & smem.pad[0].buttons
#define pad1_btn_(btn) btn & smem.pad[1].buttons
I_ B1* prim__alloc(U4 type_width, Str8 type_name) { I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
gknown PrimitiveArena* pa = & smem.primitives; gknown PrimitiveArena* pa = & smem.primitives;
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id]; gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
@@ -84,97 +92,59 @@ I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
} }
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type))) #define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue. void
* The 4 wasted-arg words for B(12h) InitPAD2 live at [SP+0..15] but are not explicitly allocated. resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
* The compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call. // RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
* // Preconditions: eye != target, up_in not collinear with (target - eye).
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers; V3_S4 right, up, forward;
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + the B-table arg registers explicitly). V3_S4 ux, uy, uz;
* The C-level writes after the call re-load the pointers from their callee-saved homes. V3_S4 pos, off;
*
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
* The kernel-ABI "volatile GPRs" subset is clb_system; the rest of the destroy set is enumerated explicitly here. */
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
{
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
(void)p0; (void)p1;
// TODO(Ed): Properly annotate the raw values in the inline asm instructions. forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
// Use enums. normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22) cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
* $a0 = raw0 (rgcc-bound; survives the sequence below) cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
* $a3 = 0x22 (immediate)
* $t1 = 0x12 (function number)
* $t2 = 0xB0 (BIOS B-table address) */
asm volatile(
asm_words(
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
add_ui( rarg_1, rdiscard, 0x22), /* $a1 = 0x22 */
add_ui( rarg_3, rdiscard, 0x22), /* $a3 = 0x22 */
add_ui( rtmp_1, rdiscard, 0x12), /* $t1 = 0x12 */
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 */
call_reg(rtmp_2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_rpins, r_use(p0), r_use(p1)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each // RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */ look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
u1_v(raw0)[0] = 0xFF; look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
u1_v(raw1)[0] = 0xFF; look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */ pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
asm volatile(
asm_words( // RGA(Lengyel): R * (-eye) is the full matrix translation column.
add_ui( rtmp_1, rdiscard, 0x13), /* $t1 = 0x13 */ // Motor translator would store half this displacement in m.xyz; GTE consumes full column.
add_ui( rtmp_2, rdiscard, 0xB0), /* $t2 = 0xB0 (re-load) */ mul_m3s2_v3s4(look_at, & pos, & off);
call_reg(rtmp_2), /* jalr $t2, $ra */ trans_m3s2( look_at, & off);
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
rlit(R_V0), rlit(R_V1),
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
rlit(R_RA),
clb_mem_drain
);
} }
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
GCC_OPTIMIZATION_DISABLE GCC_OPTIMIZATION_DISABLE
void update(PrimitiveArena* pa, U4* ordering_buf) void update(PrimitiveArena* pa, U4* ordering_buf)
{ {
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
if (1) // Pad Input // Pad Input
{ {
tb.used = 0; tb_scope_run(& tb) { tb.used = 0; tb_scope_run(& tb) {
/* BIOS-owned polling: per-frame snapshot of both ports. */ // Grab latest state from bios.
tb_emit_(pad_bios_snapshot); tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[0]); tb_data_(raw, & smem.pad_raw[0]);
tb_data_(state, & smem.pad[0]); tb_data_(state, & smem.pad[0]);
tb_emit_(pad_bios_snapshot); tb_emit_(pad_bios_snapshot);
tb_data_(raw, & smem.pad_raw[1]); tb_data_(raw, & smem.pad_raw[1]);
tb_data_(state, & smem.pad[1]); tb_data_(state, & smem.pad[1]);
/* Per-frame rotation apply: consume pad[0].buttons + pad[0].left_x */
tb_emit_(pad_apply_input); tb_emit_(pad_input_cam);
tb_data_(state, & smem.pad[0]); tb_data_(state, & smem.pad[0]);
tb_data_(cube_rot, & smem.cube.rot); tb_data_(cam, & smem.cam);
tb_data_(floor_rot, & smem.floor.rot);
// tb_emit_(pad_input_cube_rotation);
// tb_data_(state, & smem.pad[0]);
// tb_data_(cube_rot, & smem.cube.rot);
// tb_data_(floor_rot, & smem.floor.rot);
} }
} }
@@ -201,15 +171,88 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
A2_S2 p; //??? A2_S2 p; //???
S4 flag; //???? S4 flag; //????
// Camera Look at
if (0)
{
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
}
// Camera look at (Tape)
if (1)
{
MT3_S2S4* look_at = & smem.cam.look_at;
P3_S4* eye = & smem.cam.pos;
V3_S4* up_in = & v3s4(0, -fp_one, 0);
V3_S4 right, up, forward;
V3_S4 ux, uy, uz;
V3_S4 pos, off;
tb.used = 0; tb_scope_run(& tb) {
// tb_emit_bundle(resolve_look_at);
{
tb_emit_(resolve_look_at); {
tb_data_(look_at, & smem.cam.look_at);
tb_data_(eye, & smem.cam.pos);
tb_data_(target, & smem.cube.pos);
tb_data_(up_in, up_in);
// tb_emit(a_normalize_v3s4(/*Todo: resolve dependent register allocation*/));
// tb_data_(fwd_out);
}
#if 0
{
tb_emit_(resolve_look_at__resolve_right); {
//...
tb_emit_(a_normalize_v3s4(...));
tb_data_(right_out);
}
tb_emit(resolve_look_at__resolve_up); {
//...
tb_emit_(ac_normalize_v3s4(...));
tb_data_(up_out);
}
tb_emit(world_to_cam_expand_mt3_s2s4(...)); {
tb_data(look_at, & smem.cam.look_at);
}
tb_emit_(resolve_look_at__final); {
}
}
#endif
}
}
// forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
// normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
// RGA(Lengyel): R * (-eye) -- full matrix translation column.
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
mul_m3s2_v3s4(look_at, & pos, & off);
trans_m3s2( look_at, & off);
}
// Draw cube // Draw cube
if (1) if (1)
{ {
m3s2_rotation (& smem.cube.rot, & smem.tform_world); mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.cube.pos); mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
m3s2_scale (& smem.tform_world, & smem.cube.scale); mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
gte_matrix_set_rotation (& smem.tform_world);
gte_matrix_set_translation(& smem.tform_world); // Combine world and look_at matrix.
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view);
// gte_matrix_set_rotation (& smem.tform_world);
// gte_matrix_set_translation(& smem.tform_world);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]); U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used; U4 prim_cursor = prim_base + pa->used;
@@ -237,9 +280,15 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
// Draw floor // Draw floor
if (1) if (1)
{ {
m3s2_rotation (& smem.floor.rot, & smem.tform_world); mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
m3s2_translation(& smem.tform_world, & smem.floor.pos); mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
m3s2_scale (& smem.tform_world, & smem.floor.scale); mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
// Combine world and look_at matrix.
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
gte_matrix_set_rotation (& smem.tform_view);
gte_matrix_set_translation(& smem.tform_view);
U4 prim_base = u4_(pa->buf[smem.active_buf_id]); U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
U4 prim_cursor = prim_base + pa->used; U4 prim_cursor = prim_base + pa->used;
@@ -249,11 +298,11 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
// Prepare the tape. (Push protocol to tape) // Prepare the tape. (Push protocol to tape)
tb.used = 0; tb_scope(& tb) { tb.used = 0; tb_scope(& tb) {
tb_emit(& tb, set_gte_world); // tb_emit(& tb, set_gte_mt3s2s4);
tb_data(& tb, u4_(& smem.tform_world)); // tb_data(& tb, u4_(& smem.tform_view));
tb_emit(& tb, rbind_floor_f3_face); tb_emit(& tb, rbind_floor_f3_face);
// TODO(Ed): Just use a single context struct ref // TODO(Ed): Just use a single context struct ref?
tb_data(& tb, prim_cursor); tb_data(& tb, prim_cursor);
tb_data(& tb, u4_(smem.floor.faces)); tb_data(& tb, u4_(smem.floor.faces));
tb_data(& tb, u4_(smem.floor.verts)); tb_data(& tb, u4_(smem.floor.verts));
@@ -296,6 +345,7 @@ int main(void)
smem.scratchpad = C_(U4_V, 0x1F800000); smem.scratchpad = C_(U4_V, 0x1F800000);
// smem.primitives.used = 0; // smem.primitives.used = 0;
// smem.active_buf_id = 0; // smem.active_buf_id = 0;
smem.cam.pos = v3s4(500, -1000, -1500);
/*Persistent Entity Setup*/{ /*Persistent Entity Setup*/{
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); { ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
Ent_Cube* cube = & smem.cube; Ent_Cube* cube = & smem.cube;
+8 -8
View File
@@ -21,12 +21,6 @@ enum {
ScreenRes_CenterY = (ScreenRes_Y >> 1), ScreenRes_CenterY = (ScreenRes_Y >> 1),
}; };
enum {
fp_one = (1 << 12),
};
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
typedef U4 OrderingTable_Buffer[OrderingTbl_Len]; typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
typedef Array_(OrderingTable_Buffer, 2); typedef Array_(OrderingTable_Buffer, 2);
@@ -67,7 +61,7 @@ I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
typedef Struct_(Ent_Cube) { typedef Struct_(Ent_Cube) {
V3_S4 accel; V3_S4 accel;
V3_S4 vel; V3_S4 vel;
V3_S4 pos; V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S4 scale; V3_S4 scale;
V3_S2 rot; V3_S2 rot;
A8_V3_S2 verts; A8_V3_S2 verts;
@@ -94,9 +88,15 @@ I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
}; };
typedef Struct_(Ent_Floor) { typedef Struct_(Ent_Floor) {
V3_S4 accel; V3_S4 accel;
V3_S4 pos; V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S4 scale; V3_S4 scale;
V3_S2 rot; V3_S2 rot;
A4_V3_S2 verts; A4_V3_S2 verts;
A2_V3_S2 faces; A2_V3_S2 faces;
}; };
typedef Struct_(Camera) {
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
V3_S2 rot;
MT3_S2S4 look_at;
};
+5 -11
View File
@@ -180,12 +180,9 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map) $link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
$link_args += ($f_link_pass_through_prefix + $f_link_start_group) $link_args += ($f_link_pass_through_prefix + $f_link_start_group)
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim. # 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, # had LOAD lines in the map but ZERO .o files pulled in — they were unused.
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but # 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
# ZERO .o files pulled in — they were unused. The 5 kept libraries
# (api, c, etc, gpu, gte) are required by the C-side calls in
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
$libraries = @( $libraries = @(
"api", "api",
"c", "c",
@@ -227,9 +224,7 @@ function ps1-meta { param(
[string[]]$passes = @('--pre-link'), [string[]]$passes = @('--pre-link'),
[string[]]$extra_args = @() [string[]]$extra_args = @()
) )
# `--unity-root` and `--source` are # `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
# be supplied; the other must be absent.
if ($null -ne $unity_root -and $unity_root -ne '') if ($null -ne $unity_root -and $unity_root -ne '')
{ {
if ($null -ne $sources -and $sources.Count -gt 0) { if ($null -ne $sources -and $sources.Count -gt 0) {
@@ -522,7 +517,7 @@ function build-hello_camera {
$path_build_gen = join-path $path_build 'gen' $path_build_gen = join-path $path_build 'gen'
$src_c = join-path $path_module 'hello_camera.c' $src_c = join-path $path_module 'hello_camera.c'
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
$assemble_args = @() $assemble_args = @()
$assemble_args += $f_debug $assemble_args += $f_debug
@@ -557,7 +552,6 @@ function build-hello_camera {
link-modules $link_modules $elf $link_args link-modules $link_modules $elf $link_args
make-binary $elf $exe make-binary $elf $exe
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf) ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
inject-dwarf $elf $path_build_gen inject-dwarf $elf $path_build_gen
+54
View File
@@ -1053,6 +1053,8 @@ M.GTE_COMMAND_ALIASES = {
-- gte_avg_sort_z3 / gte_avg_sort_z4 are the duffle-side aliases for AVSZ3/4. -- gte_avg_sort_z3 / gte_avg_sort_z4 are the duffle-side aliases for AVSZ3/4.
["gte_avg_sort_z3"] = "gte_cmdw_avsz3", ["gte_avg_sort_z3"] = "gte_cmdw_avsz3",
["gte_avg_sort_z4"] = "gte_cmdw_avsz4", ["gte_avg_sort_z4"] = "gte_cmdw_avsz4",
["gte_cmdw_sqr"] = "gte_cmdw_sqr",
["gte_cmdw_gpf"] = "gte_cmdw_gpf",
} }
-- GTE command input-set table. -- GTE command input-set table.
@@ -1136,6 +1138,14 @@ M.GTE_COMMAND_INPUTS = {
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
"gte_cr_ZSF4", "gte_cr_ZSF4",
}, },
-- SQR: reads IR1..IR3 (per PSX-SPX gte.md SQR section; libgte disassembly 0x800160b0).
["gte_cmdw_sqr"] = {
"C2_IR1", "C2_IR2", "C2_IR3",
},
-- GPF: reads IR0 + IR1..IR3 (per PSX-SPX gte.md GPF section; libgte disassembly 0x8001613c).
["gte_cmdw_gpf"] = {
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
},
} }
-- GTE command output-set + semantic role table. -- GTE command output-set + semantic role table.
@@ -1208,6 +1218,22 @@ M.GTE_COMMAND_OUTPUTS = {
{ register = "C2_IR2", role = "latest_color" }, { register = "C2_IR2", role = "latest_color" },
{ register = "C2_IR3", role = "latest_color" }, { register = "C2_IR3", role = "latest_color" },
}, },
["gte_cmdw_sqr"] = {
{ register = "C2_MAC1", role = "mac_result" },
{ register = "C2_MAC2", role = "mac_result" },
{ register = "C2_MAC3", role = "mac_result" },
{ register = "C2_IR1", role = "latest_color" },
{ register = "C2_IR2", role = "latest_color" },
{ register = "C2_IR3", role = "latest_color" },
},
["gte_cmdw_gpf"] = {
{ register = "C2_MAC1", role = "mac_result" },
{ register = "C2_MAC2", role = "mac_result" },
{ register = "C2_MAC3", role = "mac_result" },
{ register = "C2_IR1", role = "latest_color" },
{ register = "C2_IR2", role = "latest_color" },
{ register = "C2_IR3", role = "latest_color" },
},
} }
-- GTE command/post-command latch-window table. -- GTE command/post-command latch-window table.
@@ -1270,6 +1296,22 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
{ register = "C2_IR2", required = 4 }, { register = "C2_IR2", required = 4 },
{ register = "C2_IR3", required = 4 }, { register = "C2_IR3", required = 4 },
}, },
["gte_cmdw_sqr"] = {
{ register = "C2_MAC1", required = 4 },
{ register = "C2_MAC2", required = 4 },
{ register = "C2_MAC3", required = 4 },
{ register = "C2_IR1", required = 4 },
{ register = "C2_IR2", required = 4 },
{ register = "C2_IR3", required = 4 },
},
["gte_cmdw_gpf"] = {
{ register = "C2_MAC1", required = 4 },
{ register = "C2_MAC2", required = 4 },
{ register = "C2_MAC3", required = 4 },
{ register = "C2_IR1", required = 4 },
{ register = "C2_IR2", required = 4 },
{ register = "C2_IR3", required = 4 },
},
} }
-- Operand-class table for the COP2->GPR load-delay check. -- Operand-class table for the COP2->GPR load-delay check.
@@ -1285,6 +1327,7 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
M.OPERAND_READ_POSITIONS = { M.OPERAND_READ_POSITIONS = {
-- CPU ALU with one or two GPR operands. Reads every GPR operand. -- CPU ALU with one or two GPR operands. Reads every GPR operand.
["add_ui"] = {1, 2}, ["add_ui"] = {1, 2},
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
["add_ui_self"] = {1}, ["add_ui_self"] = {1},
["add_si"] = {1, 2}, ["add_si"] = {1, 2},
["add_u"] = {1, 2, 3}, ["add_u"] = {1, 2, 3},
@@ -1354,6 +1397,8 @@ M.OPERAND_READ_POSITIONS = {
["gte_mv_to_ctrl_r"] = {}, ["gte_mv_to_ctrl_r"] = {},
["gte_lw"] = {}, ["gte_lw"] = {},
["gte_sw"] = {}, ["gte_sw"] = {},
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
["shift_aright_var"] = {1, 2, 3},
} }
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte. -- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
@@ -1435,8 +1480,10 @@ M.INSTRUCTION_LATENCY = {
["xor_i"] = 1, ["xor_u"] = 1, ["xor_i"] = 1, ["xor_u"] = 1,
["nor_u"] = 1, ["nor_u"] = 1,
["shift_lleft"] = 1, ["shift_lleft_self"] = 1, ["shift_lleft"] = 1, ["shift_lleft_self"] = 1,
["shift_lleft_var"] = 1, -- sllv: 1 cycle
["shift_lright"] = 1, ["shift_lright"] = 1,
["shift_aright"] = 1, ["shift_aright"] = 1,
["shift_aright_var"] = 1, -- srav: 1 cycle
["mask_upper"] = 1, ["mask_upper"] = 1,
["mov_from_high"] = 2, -- mfhi: 2 cycles ["mov_from_high"] = 2, -- mfhi: 2 cycles
["mov_from_low"] = 2, -- mflo: 2 cycles ["mov_from_low"] = 2, -- mflo: 2 cycles
@@ -1454,6 +1501,7 @@ M.INSTRUCTION_LATENCY = {
["load_half_u"] = 1, ["load_half"] = 1, ["load_half_u"] = 1, ["load_half"] = 1,
["load_byte_u"] = 1, ["load_byte"] = 1, ["load_byte_u"] = 1, ["load_byte"] = 1,
["load_upper_i"] = 1, ["load_upper_i"] = 1,
["li_s"] = 1, -- aliased to add_ui(rt, R_0, imm); 1 cycle
-- 2-word loads (lui + ori) used for >16-bit immediates -- 2-word loads (lui + ori) used for >16-bit immediates
["load_imm"] = 2, ["load_imm"] = 2,
["load_imm_1w"] = 1, ["load_imm_1w"] = 1,
@@ -1497,6 +1545,8 @@ M.INSTRUCTION_LATENCY = {
["gte_cmdw_op"] = 6, -- OP: 6 cycles (PSX-SPX) ["gte_cmdw_op"] = 6, -- OP: 6 cycles (PSX-SPX)
["gte_cmdw_outer_product"] = 6, -- alias for OP ["gte_cmdw_outer_product"] = 6, -- alias for OP
["gte_cmdw_wedge"] = 6, -- alias for OP ["gte_cmdw_wedge"] = 6, -- alias for OP
["gte_cmdw_sqr"] = 5, -- SQR(sf): 5 cycles (PSX-SPX); +2 nops for pre-fill if sf=0/1
["gte_cmdw_gpf"] = 5, -- GPF(sf,lm): 5 cycles (PSX-SPX); +2 nops for pre-fill if needed
-- Long-form aliases (same cycle cost as their short form) -- Long-form aliases (same cycle cost as their short form)
["gte_cmdw_rotate_translate_perspective_single"] = 15, -- alias for rtps ["gte_cmdw_rotate_translate_perspective_single"] = 15, -- alias for rtps
["gte_cmdw_rotate_translate_perspective_triple"] = 23, -- alias for rtpt ["gte_cmdw_rotate_translate_perspective_triple"] = 23, -- alias for rtpt
@@ -1777,6 +1827,7 @@ M.CU2_TRANSITION_POLICY = {
M.INSTRUCTION_GPR_EFFECTS = { M.INSTRUCTION_GPR_EFFECTS = {
-- CPU ALU with one or two GPR operands. Reads every GPR operand position. -- CPU ALU with one or two GPR operands. Reads every GPR operand position.
add_ui = { reads = {1, 2}, writes = {1} }, add_ui = { reads = {1, 2}, writes = {1} },
li_s = { reads = {1, 2}, writes = {1} }, -- RMW: rt is both read + written
add_ui_self = { reads = {1}, writes = {1} }, add_ui_self = { reads = {1}, writes = {1} },
add_si = { reads = {1, 2}, writes = {1} }, add_si = { reads = {1, 2}, writes = {1} },
add_u = { reads = {2, 3}, writes = {1} }, add_u = { reads = {2, 3}, writes = {1} },
@@ -1893,6 +1944,8 @@ M.INSTRUCTION_GPR_EFFECTS = {
atom_writes = { reads = {}, writes = {} }, atom_writes = { reads = {}, writes = {} },
-- mac_yield transfers control to the next atom; zero GPR effects. -- mac_yield transfers control to the next atom; zero GPR effects.
mac_yield = { reads = {}, writes = {} }, mac_yield = { reads = {}, writes = {} },
shift_lleft_var = { reads = {2, 3}, writes = {1} },
shift_aright_var = { reads = {2, 3}, writes = {1} },
} }
-- Bounded GPR-value rules consumed by the same forward event walk as `INSTRUCTION_GPR_EFFECTS`. -- Bounded GPR-value rules consumed by the same forward event walk as `INSTRUCTION_GPR_EFFECTS`.
@@ -1905,6 +1958,7 @@ M.INSTRUCTION_GPR_EFFECTS = {
M.GPR_VALUE_RULES = { M.GPR_VALUE_RULES = {
load_upper_i = { op = "load_upper_i", dest = 1, immediate = 2, }, load_upper_i = { op = "load_upper_i", dest = 1, immediate = 2, },
add_ui = { op = "add_ui", dest = 1, source = 2, immediate = 3, }, add_ui = { op = "add_ui", dest = 1, source = 2, immediate = 3, },
li_s = { op = "add_ui", dest = 1, source = 2, immediate = 3 }, -- R_0 + sign-ext(imm) folds into a constant
or_i = { op = "or_i", dest = 1, source = 2, immediate = 3, }, or_i = { op = "or_i", dest = 1, source = 2, immediate = 3, },
and_i = { op = "and_i", dest = 1, source = 2, immediate = 3, }, and_i = { op = "and_i", dest = 1, source = 2, immediate = 3, },
xor_i = { op = "xor_i", dest = 1, source = 2, immediate = 3, }, xor_i = { op = "xor_i", dest = 1, source = 2, immediate = 3, },
+418
View File
@@ -0,0 +1,418 @@
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
-- both parsed ELF32 headers from wire bytes.
--
-- This module contains the format constants and the byte-level walker.
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
--
-- **Adapter contract (explicit pass style):**
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
-- so callers wrap once in a 1-line adapter that strips `self`.
-- The parsers here operate on the unwrapped form.
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
-- read_u8(adapter, off) -> integer | nil
-- read_u16(adapter, off) -> integer | nil
-- read_u32(adapter, off) -> integer | nil
-- size(adapter) -> integer
--
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
--
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
local M = {}
-- ════════════════════════════════════════════════════════════════════════════
-- Little-endian readers (bit-weighted accumulator, math.floor only)
-- ════════════════════════════════════════════════════════════════════════════
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
---
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
--- byte 0 contributes its value directly;
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
---
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
---
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u32(adapter, off)
return adapter.read_u8_at(off)
+ adapter.read_u8_at(off + 0x01) * 0x00000100
+ adapter.read_u8_at(off + 0x02) * 0x00010000
+ adapter.read_u8_at(off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u16(adapter, off)
return adapter.read_u8_at(off)
+ adapter.read_u8_at(off + 0x01) * 0x00000100
end
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
--- @param adapter table
--- @param off integer -- zero-based wire offset
--- @return integer|nil
function M.read_u8(adapter, off)
return adapter.read_u8_at(off)
end
--- Total adapter byte length.
--- @param adapter table
--- @return integer
function M.size(adapter)
return adapter.read_size()
end
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
--- both layers now use the same byte-level helpers under the hood.
function M.read_u32_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off)
+ buf:byte(byte_off + 0x01) * 0x00000100
+ buf:byte(byte_off + 0x02) * 0x00010000
+ buf:byte(byte_off + 0x03) * 0x01000000
end
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- @param buf string
--- @param off integer -- zero-based wire offset
--- @return integer
function M.read_u16_le(buf, off)
local byte_off = off + 1
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
end
-- ════════════════════════════════════════════════════════════════════════════
-- Format constants
-- ════════════════════════════════════════════════════════════════════════════
-- ELF format constants (System V ABI gABI v1.2).
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
-- ---------------------------------------------------------------------------
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
-- ---------------------------------------------------------------------------
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
M.ELF32_HEADER = {
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
magic = "\127ELF",
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
header_bytes = 0x34, -- ELF32 header is 52 bytes total
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
}
-- ---------------------------------------------------------------------------
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
-- ---------------------------------------------------------------------------
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
-- zero-based, field offsets relative to the start of the entry.
M.ELF32_SECTION = {
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
}
-- ---------------------------------------------------------------------------
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
-- ---------------------------------------------------------------------------
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
-- zero-based, field offsets relative to the start of the entry.
M.ELF32_SYM = {
st_name = 0x00, -- 4-byte LE; offset into the linked string table
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
st_size = 0x08, -- 4-byte LE; symbol size in bytes
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
}
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
M.dw_dwarf32_terminator = 0xFFFFFFFF
-- ════════════════════════════════════════════════════════════════════════════
-- Adapter validation
-- ════════════════════════════════════════════════════════════════════════════
--- Validate that `adapter` exposes the byte-read surface.
--- Returns true on success, false + a stable error code on failure.
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
--- @param adapter any
--- @return boolean, string|nil
function M.validate_adapter(adapter)
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
return true, nil
end
-- ════════════════════════════════════════════════════════════════════════════
-- String-table reader
-- ════════════════════════════════════════════════════════════════════════════
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
--- @param strtab string
--- @param off integer
--- @return string|nil
function M.get_str(strtab, off)
if off < 0 or off >= #strtab then return nil end
local end_pos = strtab:find("\0", off + 1, true)
if not end_pos then return nil end
return strtab:sub(off + 1, end_pos - 1)
end
-- ════════════════════════════════════════════════════════════════════════════
-- Header / section / symbol walkers
-- ════════════════════════════════════════════════════════════════════════════
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
--- Returns a table on success:
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
--- On failure returns nil + a stable error code:
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
--- @param adapter table
--- @return table|nil, string|nil
function M.parse_elf32_headers(adapter)
local ok, err = M.validate_adapter(adapter)
if not ok then return nil, err end
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
-- The byte readers take the adapter explicitly.
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
local b1 = M.read_u8(adapter, 0)
local b2 = M.read_u8(adapter, 1)
local b3 = M.read_u8(adapter, 2)
local b4 = M.read_u8(adapter, 3)
if not (b1 and b2 and b3 and b4)
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
return nil, "bad_magic"
end
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset)
if class ~= M.ELFCLASS32 then
return nil, "unsupported_elf_class"
end
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset)
if data ~= M.ELFDATA2LSB then
return nil, "unsupported_elf_data"
end
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset)
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset)
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset)
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset)
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset)
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
return nil, "truncated_header"
end
return {
e_entry = e_entry,
e_shoff = e_shoff,
e_shentsize = e_shentsize,
e_shnum = e_shnum,
e_shstrndx = e_shstrndx,
error = nil,
}
end
--- Read one section-header entry from `adapter` at `sh_off`.
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
--- @param adapter table
--- @param sh_off integer
--- @return table|nil, string|nil -- entry, error
local function read_section_entry(adapter, sh_off)
local entry = {
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
name = "",
}
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
and entry.sh_offset and entry.sh_size and entry.sh_link) then
return nil, "truncated_section_headers"
end
return entry, nil
end
--- Walk every section header in `hdr` and return a 1-based array of entries
--- (the section at logical index 0 is at array position 1, etc.).
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
--- @param adapter table
--- @param hdr table -- the table returned by parse_elf32_headers
--- @return table|nil, string|nil
function M.walk_sections(adapter, hdr)
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
local file_size = M.size(adapter)
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
return nil, "truncated_section_headers"
end
-- Read every section header first; we need .shstrtab to resolve names.
local sections = {}
for i = 0, hdr.e_shnum - 1 do
local sh_off = hdr.e_shoff + i * hdr.e_shentsize
local entry, err = read_section_entry(adapter, sh_off)
if not entry then return nil, err end
sections[i + 1] = entry
end
if hdr.e_shstrndx >= hdr.e_shnum then
return nil, "missing_shstrtab"
end
local shstrtab = sections[hdr.e_shstrndx + 1]
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
return nil, "missing_shstrtab"
end
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
return nil, "truncated_section_headers"
end
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab)
if not shstrtab_bytes then return nil, "truncated_section_headers" end
for _, s in ipairs(sections) do
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
end
return sections, nil
end
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
--- @param adapter table
--- @param section table -- one entry from walk_sections
--- @return string|nil
function M.read_section_bytes(adapter, section)
local size = section.sh_size
if size == 0 then return "" end
local out = {}
for i = 0, size - 1 do
local b = M.read_u8(adapter, section.sh_offset + i)
if b == nil then return nil end
out[#out + 1] = string.char(b)
end
return table.concat(out)
end
--- Convenience: walk sections, then look up the named section, then read its bytes.
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
--- @param adapter table
--- @param sections table -- 1-based array from walk_sections
--- @param name string
--- @return string|nil, string|nil
function M.read_named_section(adapter, sections, name)
if not sections then return nil, "missing_section" end
for _, s in ipairs(sections) do
if s.name == name then
local bytes = M.read_section_bytes(adapter, s)
if not bytes then return nil, "truncated_section_data" end
return bytes, nil
end
end
return nil, "missing_section"
end
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
--- @param adapter table
--- @param sections table
--- @return table|nil, string|nil
function M.collect_symbols(adapter, sections)
if not sections then return nil, "missing_sections" end
local symbols = {}
local file_size = M.size(adapter)
for _, s in ipairs(sections) do
if s.sh_type == M.SHT_SYMTAB then
local strtab = sections[s.sh_link + 1]
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
return nil, "missing_symtab_strtab"
end
if strtab.sh_offset + strtab.sh_size > file_size then
return nil, "truncated_section_headers"
end
local strtab_bytes = M.read_section_bytes(adapter, strtab)
if not strtab_bytes then return nil, "truncated_section_headers" end
if s.sh_offset + s.sh_size > file_size then
return nil, "truncated_section_headers"
end
local symtab_bytes = M.read_section_bytes(adapter, s)
if not symtab_bytes then return nil, "truncated_section_headers" end
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes
for j = 0, n - 1 do
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name)
if st_name then
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value)
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size)
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info)
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
-- the metaprogram reads too. Inline the read to keep the
-- adapter as the only I/O surface.
local b1 = M.read_u8(adapter, e + 14)
local b2 = M.read_u8(adapter, e + 15)
if not (b1 and b2) then
return nil, "truncated_section_headers"
end
local st_shndx = b1 + b2 * 0x100
local name = M.get_str(strtab_bytes, st_name) or ""
if name ~= "" then
symbols[name] = {
value = st_value,
size = st_size,
info = st_info,
shndx = st_shndx,
}
end
end
end
end
end
return symbols, nil
end
return M
+150 -137
View File
@@ -11,6 +11,11 @@
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`). -- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
local lfs = require("lfs") local lfs = require("lfs")
-- scripts/elf32.lua contains format-constant tables + the byte-level walker.
-- The this file re-exports `read_u32_le` / `read_u16_le` (and the DWARF32 terminator).
-- TODO(Ed): Remove re-export.
local E = require("elf32")
local M = {} local M = {}
-- ════════════════════════════════════════════════════════════════════════════ -- ════════════════════════════════════════════════════════════════════════════
@@ -102,27 +107,13 @@ M.MIPS_BYTES_PER_WORD = 0x04
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets. --- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
--- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`). --- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
--- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file. --- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
---
--- The ELF32 header / section / sym layout tables are within scripts/elf32.lua.
--- The metaprogram re-exports the DWARF32 initial-length terminator.
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table" --- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
M.ELF32 = { M.dw_dwarf32_terminator = E.dw_dwarf32_terminator
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00 -- TODO(Ed): Remove re-export.
magic = "\127ELF",
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
class_elf32 = 1,
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
endian_little = 1,
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
}
-- ---------------------------------------------------------------------------- -- ----------------------------------------------------------------------------
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table) -- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
@@ -241,27 +232,24 @@ M.DWARF5_DEBUG_LINE = {
--- (which has partial `string.unpack` coverage). --- (which has partial `string.unpack` coverage).
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary. --- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
--- ---
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit: --- Thin forwarder: the canonical implementation lives in scripts/elf32.lua.
--- byte 0 contributes its value directly; byte 1 is shifted left by 8 (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000). --- The "second caller lifts" pattern keeps the metaprogram side fluent
--- (`M.read_u32_le(buf, off)`) while the body is deduped.
--- @param buf string --- @param buf string
--- @param off integer -- zero-based wire offset --- @param off integer -- zero-based wire offset
--- @return integer --- @return integer
function M.read_u32_le(buf, off) function M.read_u32_le(buf, off)
local byte_off = off + 1 return E.read_u32_le(buf, off)
return buf:byte(byte_off)
+ buf:byte(byte_off + 0x01) * 0x00000100
+ buf:byte(byte_off + 0x02) * 0x00010000
+ buf:byte(byte_off + 0x03) * 0x01000000
end end
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`. --- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.) --- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
--- Thin forwarder — see `M.read_u32_le` for the rationale.
--- @param buf string --- @param buf string
--- @param off integer -- zero-based wire offset --- @param off integer -- zero-based wire offset
--- @return integer --- @return integer
function M.read_u16_le(buf, off) function M.read_u16_le(buf, off)
local byte_off = off + 1 return E.read_u16_le(buf, off)
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
end end
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser. -- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
@@ -442,20 +430,20 @@ function M.read_ref_sig8(buf, pos)
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8 return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
end end
-- DWARF5 §7.5.6 (Type Entries). --- DWARF5 §7.5.6 (Type Entries).
-- Walk all units in `info` and return the 0-based offset of the first unit --- Walk all units in `info` and return the 0-based offset of the first unit whose `DW_AT_type_signature`
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`. --- (8-byte value at the end of the unit header) equals `target_sig`.
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract; --- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists. --- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
-- ---
-- Unit header layout (from pos 0): --- Unit header layout (from pos 0):
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4) --- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4) --- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset). --- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
-- @param info string -- the .debug_info section bytes --- @param info string -- the .debug_info section bytes
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature --- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature --- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit --- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi) function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
local pos = 0 local pos = 0
local section_len = #info local section_len = #info
@@ -564,69 +552,58 @@ function M.read_elf_sections(elf_path, section_names)
return result return result
end end
-- Read the ELF32 header. local file_size
local header = f:read(M.ELF32.header_bytes) do
if not header or #header < M.ELF32.header_bytes then f:seek("end", 0)
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n") file_size = f:seek("cur", 0)
end
local adapter = {
read_u8_at = function(offset)
f:seek("set", offset)
local b = f:read(1)
if not b then return nil end
return b:byte()
end,
read_u16_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
if not b1 or not b2 then return nil end
return b1:byte() + b2:byte() * 0x100
end,
read_u32_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
local b3 = f:read(1)
local b4 = f:read(1)
if not b1 or not b2 or not b3 or not b4 then return nil end
return b1:byte() + b2:byte() * 0x100
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
end,
read_size = function() return file_size end,
}
-- Delegate the header parse + section walk to E.*.
local hdr, hdr_err = E.parse_elf32_headers(adapter)
if not hdr then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] header parse failed: %s\n", tostring(hdr_err)))
f:close() f:close()
return result return result
end end
-- Sanity-check magic + class + endianness. local sections, walk_err = E.walk_sections(adapter, hdr)
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then if not sections then
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n") io.stderr:write(string.format("[elf_dwarf.read_elf_sections] section walk failed: %s\n", tostring(walk_err)))
f:close()
return result
end
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
f:close()
return result
end
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
f:close() f:close()
return result return result
end end
-- Parse section-header table location + dimensions from the header. -- Resolve the requested sections.
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset) for _, s in ipairs(sections) do
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset) if wanted[s.name] then
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset) local bytes = E.read_section_bytes(adapter, s)
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset) if bytes then result[s.name] = bytes end
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
local strtab_hdr = f:read(e_shentsize)
if not strtab_hdr or #strtab_hdr < e_shentsize then
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
f:close()
return result
end
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
f:seek("set", strtab_offset)
local strtab = f:read(strtab_size) or ""
-- Walk all section headers; collect (offset, size) for the wanted names.
local function read_section_bytes(sh_offset, sh_size)
f:seek("set", sh_offset)
return f:read(sh_size) or ""
end
for sh_idx = 0, e_shnum - 1 do
f:seek("set", e_shoff + sh_idx * e_shentsize)
local sh = f:read(e_shentsize)
if not sh or #sh < e_shentsize then break end
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
-- Extract the name (null-terminated C string in strtab).
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
local name = strtab:sub(sh_name + 1, name_end - 1)
if wanted[name] then
result[name] = read_section_bytes(sh_offset, sh_size)
end end
end end
@@ -643,48 +620,87 @@ end
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded. --- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix). --- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
--- - `st_size > 0` filter excludes undefined/imported symbols. --- - `st_size > 0` filter excludes undefined/imported symbols.
---
--- @param elf_path Path --- @param elf_path Path
--- @return table<string, {integer, integer}> --- @return table<string, {integer, integer}>
function M.read_nm(elf_path) function M.read_nm(elf_path)
local addrs = {} local addrs = {}
-- Read .symtab + .strtab via the existing ELF walker (no subprocess). -- Existence check first; an empty or missing ELF returns an empty map.
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"}) if lfs.attributes(elf_path, "mode") ~= "file" then
local symtab = sections[".symtab"]
local strtab = sections[".strtab"]
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
-- No symbol table (e.g. stripped ELF). Return empty.
return addrs return addrs
end end
-- Iterate the 16-byte ELF32 symtab entries. local f = io.open(elf_path, "rb")
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14. if not f then
local SYM_ENTRY_BYTES = 0x10 return addrs
local SYM_ST_NAME = 0x00
local SYM_ST_VALUE = 0x04
local SYM_ST_SIZE = 0x08
local SYM_ST_INFO = 0x0C
local n_syms = #symtab / SYM_ENTRY_BYTES
for i = 0, n_syms - 1 do
local entry_off = i * SYM_ENTRY_BYTES
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1)
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions).
local binding = math.floor(st_info / 16)
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE)
if st_size > 0 then
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME)
-- Extract the name from .strtab (null-terminated C string).
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
local name = strtab:sub(st_name_off + 1, name_end - 1)
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
if name and #name > 0 then
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
addrs[name] = { st_value, st_size }
end end
-- Build the file adapter for E.*.
local file_size
do
f:seek("end", 0)
file_size = f:seek("cur", 0)
end end
local adapter = {
read_u8_at = function(offset)
f:seek("set", offset)
local b = f:read(1)
if not b then return nil end
return b:byte()
end,
read_u16_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
if not b1 or not b2 then return nil end
return b1:byte() + b2:byte() * 0x100
end,
read_u32_at = function(offset)
f:seek("set", offset)
local b1 = f:read(1)
local b2 = f:read(1)
local b3 = f:read(1)
local b4 = f:read(1)
if not b1 or not b2 or not b3 or not b4 then return nil end
return b1:byte() + b2:byte() * 0x100
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
end,
read_size = function() return file_size end,
}
-- Delegate the header + section walk to E.*.
local hdr, hdr_err = E.parse_elf32_headers(adapter)
if not hdr then
io.stderr:write(string.format("[elf_dwarf.read_nm] header parse failed: %s\n", tostring(hdr_err)))
f:close()
return addrs
end
local sections, walk_err = E.walk_sections(adapter, hdr)
if not sections then
io.stderr:write(string.format("[elf_dwarf.read_nm] section walk failed: %s\n", tostring(walk_err)))
f:close()
return addrs
end
-- E.collect_symbols returns every defined symbol (no binding filter).
-- The metaprogram then applies its STB_LOCAL / STB_GLOBAL + size>0 filter, matching `nm`'s default (external symbols only).
local symbols, sym_err = E.collect_symbols(adapter, sections)
if not symbols then
io.stderr:write(string.format("[elf_dwarf.read_nm] symbol collection failed: %s\n", tostring(sym_err)))
f:close()
return addrs
end
f:close()
for name, entry in pairs(symbols) do
-- High nibble of st_info = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
-- math.floor(/16) is portable across LuaJIT 2.0/2.1 and plain Lua 5.x.
local binding = math.floor(entry.info / 16)
if (binding == 0 or binding == 1) and entry.size > 0 then
addrs[name] = { entry.value, entry.size }
end end
end end
@@ -822,12 +838,11 @@ end
--- * The `.debug_line` section may contain MULTIPLE line-program units --- * The `.debug_line` section may contain MULTIPLE line-program units
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc. --- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program) --- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom --- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom `inv.call_file`
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`. --- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH. --- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` --- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` emits 2 forms: path + dir_index).
--- emits 2 forms: path + dir_index). The helper supports: --- The helper supports:
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str) --- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line) --- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
--- - DW_FORM_udata (ULEB128) --- - DW_FORM_udata (ULEB128)
@@ -837,9 +852,7 @@ end
--- ---
--- Behavior on failure: writes to stderr and returns nil. --- Behavior on failure: writes to stderr and returns nil.
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map; --- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
--- downstream `resolve_provenance_file_index(path)` consumers --- downstream `resolve_provenance_file_index(path)` consumers consult the map directly.
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
--- consult the map directly.
--- ---
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`) --- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
--- @return table|nil, table|nil, table|nil --- @return table|nil, table|nil, table|nil
+3 -1
View File
@@ -299,7 +299,9 @@ local function word_count_rec(name, comp_by_name, wc, cache)
local trimmed = t.tok local trimmed = t.tok
if trimmed ~= "" then if trimmed ~= "" then
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1)) local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
if lookup and comp_by_name[lookup] then if lookup == "atom_label" or lookup == "atom_offset" then
-- Pure metaprogram anchors; emit zero words.
elseif lookup and comp_by_name[lookup] then
-- It's a `mac_X(...)` call. Recurse. -- It's a `mac_X(...)` call. Recurse.
n = n + word_count_rec(lookup, comp_by_name, wc, cache) n = n + word_count_rec(lookup, comp_by_name, wc, cache)
elseif lookup and wc and wc[lookup] then elseif lookup and wc and wc[lookup] then
+2 -2
View File
@@ -992,7 +992,7 @@ local function build_dwarf_line_section(existing, atom_table)
while unit_pos < #existing do while unit_pos < #existing do
if unit_pos + 4 > #existing then return existing end if unit_pos + 4 > #existing then return existing end
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos) local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
if unit_length == elf_dwarf.ELF32.dw_dwarf32_terminator then return existing end if unit_length == elf_dwarf.dw_dwarf32_terminator then return existing end
local unit_end_excl = unit_pos + 4 + unit_length local unit_end_excl = unit_pos + 4 + unit_length
if unit_end_excl > #existing then return existing end if unit_end_excl > #existing then return existing end
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
@@ -1053,7 +1053,7 @@ local function build_dwarf_aranges_section(existing, atom_table)
while i < #existing do while i < #existing do
-- Read this unit's length. -- Read this unit's length.
local ul = elf_dwarf.read_u32_le(existing, i) local ul = elf_dwarf.read_u32_le(existing, i)
if ul == elf_dwarf.ELF32.dw_dwarf32_terminator then if ul == elf_dwarf.dw_dwarf32_terminator then
-- DWARF64 marker - not supported. -- DWARF64 marker - not supported.
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n") io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
return existing return existing
+76 -21
View File
@@ -256,8 +256,21 @@ local BRANCH_PATTERN = "^branch_[%w_]+%s*%("
-- The C preprocessor expands it BEFORE the metaprogram sees the source, but for source-level metadata consistency we still match it here and classify it as a branch_equal. -- The C preprocessor expands it BEFORE the metaprogram sees the source, but for source-level metadata consistency we still match it here and classify it as a branch_equal.
-- This keeps `consuming_encoder` canonical for any downstream tooling that consults the metadata field. -- This keeps `consuming_encoder` canonical for any downstream tooling that consults the metadata field.
local JUMP_REL_PATTERN = "^jump_rel%s*%(" local JUMP_REL_PATTERN = "^jump_rel%s*%("
local UNCOND_JUMP_PATTERN = "^%f[%w](jump|call_addr)%f[%W]" local UNCOND_JUMP_PATTERNS = {
local TERMINAL_JUMP_PATTERN = "^%f[%w](jump_reg|call_reg|jump_link)%f[%W]" "^%f[%w]jump%f[%W]",
"^%f[%w]call_addr%f[%W]",
}
local TERMINAL_JUMP_PATTERNS = {
"^%f[%w]jump_reg%f[%W]",
"^%f[%w]call_reg%f[%W]",
"^%f[%w]jump_link%f[%W]",
}
local function matches_any(tok, patterns)
for i = 1, #patterns do
if tok:match(patterns[i]) then return true end
end
return false
end
local function classify_tokens(tokens) local function classify_tokens(tokens)
local n = #tokens local n = #tokens
@@ -301,13 +314,13 @@ local function classify_tokens(tokens)
-- Both encode a 16-bit signed relative word offset. -- Both encode a 16-bit signed relative word offset.
is_branch = true is_branch = true
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
elseif tok:match(UNCOND_JUMP_PATTERN) then elseif matches_any(tok, UNCOND_JUMP_PATTERNS) then
-- Unconditional absolute jump / call: `jump(off)` / `call_addr(off)`. -- Unconditional absolute jump / call: `jump(off)` / `call_addr(off)`.
-- One immediate offset field; can carry an `atom_offset(F, T)` marker (the offsets pass dispatches on `consuming_encoder` — see `passes/offsets.lua::compute_offsets`). -- One immediate offset field; can carry an `atom_offset(F, T)` marker (the offsets pass dispatches on `consuming_encoder` — see `passes/offsets.lua::compute_offsets`).
is_branch = true is_branch = true
is_unconditional_jump = true is_unconditional_jump = true
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
elseif tok:match(TERMINAL_JUMP_PATTERN) then elseif matches_any(tok, TERMINAL_JUMP_PATTERNS) then
-- Register-form jump / call: no offset field; `atom_offset` is invalid here (the offsets pass will error if one is supplied). -- Register-form jump / call: no offset field; `atom_offset` is invalid here (the offsets pass will error if one is supplied).
-- Transfers control OUT of the current atom — the CFG treats this as a path terminator. -- Transfers control OUT of the current atom — the CFG treats this as a path terminator.
is_terminal_jump = true is_terminal_jump = true
@@ -567,13 +580,31 @@ local function evaluate_gpr_value_rule(rule, ev_args, gpr_values)
return shift_left_u4(immediate % 0x10000, 16) return shift_left_u4(immediate % 0x10000, 16)
end end
local source = nil -- Encoders that take `R_0` implicitly (e.g. `li_s(rt, imm)` which is `add_ui(rt, R_0, imm)`) have a non-GPR operand at the source position.
-- Fall back to R_0 = 0.
-- The implicit-R_0 macros also use a different immediate position (e.g. `li_s`'s `add_ui` rule has source = 2 / immediate = 3
-- but the macro takes 2 args); when the configured immediate position is out of bounds.
-- Fall back instead to scanning the macro's args for the first integer literal and use that as the immediate.
local source = 0
if rule.source then if rule.source then
if is_gpr_operand(ev_args[rule.source]) then
source = constant_for_operand(gpr_values, ev_args[rule.source]) source = constant_for_operand(gpr_values, ev_args[rule.source])
if source == nil then return nil end if source == nil then return nil end
end end
local immediate = rule.immediate and parse_integer_literal(ev_args[rule.immediate]) or nil -- Non-GPR at source position = implicit R_0; source stays 0.
if rule.immediate and immediate == nil then return nil end end
local immediate = nil
if rule.immediate and ev_args[rule.immediate] ~= nil then
immediate = parse_integer_literal(ev_args[rule.immediate])
if immediate == nil then return nil end
elseif rule.immediate then
-- Immediate position out of bounds: scan for the first integer literal in the args.
for _, arg in ipairs(ev_args) do
immediate = parse_integer_literal(arg)
if immediate ~= nil then break end
end
if immediate == nil then return nil end
end
if operation == "add_ui" then return wrap_u4( source + sign_extend_i16(immediate)) if operation == "add_ui" then return wrap_u4( source + sign_extend_i16(immediate))
elseif operation == "or_i" then return bit_binary( source, immediate % 0x10000, "or") elseif operation == "or_i" then return bit_binary( source, immediate % 0x10000, "or")
elseif operation == "and_i" then return bit_binary( source, immediate % 0x10000, "and") elseif operation == "and_i" then return bit_binary( source, immediate % 0x10000, "and")
@@ -1433,17 +1464,20 @@ end
--- The register becomes non-volatile again at word N+2 (the load has retired), OR sooner if a non-load instruction overwrites the register --- The register becomes non-volatile again at word N+2 (the load has retired), OR sooner if a non-load instruction overwrites the register
--- (the overwriter's write is the fresh producer; the load's value is shadowed and never observed by any reader). --- (the overwriter's write is the fresh producer; the load's value is shadowed and never observed by any reader).
--- ---
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt: their internal load-then-use sequences --- Runtime-helper atoms / components (`debug_skip == true`) are exempt from some checks, but load-delay
--- are part of the fixed handshake (e.g. `ac_load_tri_indices` loads into R_T0..R_T2, but those are caller-supplied). --- safety applies to their emitted instructions as well.
--- ---
--- The walker reads `duffle.OPERAND_READ_POSITIONS[event.encoder]` to determine which args are read-source --- The walker reads `duffle.OPERAND_READ_POSITIONS[event.encoder]` to determine which args are read-source
--- (the destination of a load is in `writes`, not `reads` — see `duffle.INSTRUCTION_GPR_EFFECTS`). --- (the destination of a load is in `writes`, not `reads` — see `duffle.INSTRUCTION_GPR_EFFECTS`).
--- The check is purely structural; it does not consult the GPR-value lattice (no constant propagation needed for load-delay detection — the volatility window is unconditional). --- The check is purely structural; it does not consult the GPR-value lattice
--- (no constant propagation needed for load-delay detection — the volatility window is unconditional).
local function check_load_delay_slots(atom, pipe_ctx, findings) local function check_load_delay_slots(atom, pipe_ctx, findings)
-- The load-delay check applies to every atom and component body, including debug-skipped components (`ac_*` and `atom_dbg_skip MipsAtom_(...)`).
-- The `atom_dbg_skip` marker controls debugger stepping, not instruction safety.
local p = atom.paths or {}
if atom.kind ~= "atom" then return end if atom.kind ~= "atom" then return end
local events = atom.paths.word_events or {} local events = p.word_events or {}
if #events == 0 then return end if #events == 0 then return end
if is_runtime_helper(atom) then return end
local gpr_effects = duffle.INSTRUCTION_GPR_EFFECTS or {} local gpr_effects = duffle.INSTRUCTION_GPR_EFFECTS or {}
local read_positions = duffle.OPERAND_READ_POSITIONS or {} local read_positions = duffle.OPERAND_READ_POSITIONS or {}
@@ -1656,24 +1690,42 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
return atom.line + line_in_body[tokens[idx].rel] return atom.line + line_in_body[tokens[idx].rel]
end end
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot. -- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s (natural fall-through load pattern).
-- When the pattern is satisfied, the check stays silent; only violations emit findings.
for tok_idx = 1, n do for tok_idx = 1, n do
local c = tc[tok_idx] local c = tc[tok_idx]
if c.ident == "mac_yield_load" then if c.ident == "mac_yield_load" then
if tok_idx < 2 or not tc[tok_idx - 1].is_branch then local prev_tc = (tok_idx >= 2) and tc[tok_idx - 1] or nil
local prev_ident = (tok_idx >= 2) and (tc[tok_idx - 1].ident or "?") or "<none>" -- Look for the next `atom_label()` token (skip `atom_offset` markers; check immediately-adjacent first).
local next_label_tc = (tok_idx + 1 <= n) and tc[tok_idx + 1] or nil
if next_label_tc and next_label_tc.ident ~= "atom_label" then
next_label_tc = nil
for j = tok_idx + 1, n do
local t = tc[j]
if t.ident == "atom_label" then
next_label_tc = t
break
end
end
end
local natural_fallthrough = prev_tc and prev_tc.is_atom_label and next_label_tc ~= nil
if not natural_fallthrough then
if tok_idx < 2 or not prev_tc.is_branch then
local prev_ident = prev_tc and (prev_tc.ident or "?") or "<none>"
local next_ident = next_label_tc and (next_label_tc.ident .. "(" .. (next_label_tc.label_name or "?") .. ")") or "<no following label>"
findings[#findings + 1] = { findings[#findings + 1] = {
atom = atom.name, atom = atom.name,
line = tok_idx >= 2 and line_for(tok_idx) or atom.line, line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
check = "yield_load_tail_pairing", check = "yield_load_tail_pairing",
kind = "error", kind = "error",
msg = string.format( msg = string.format(
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — `mac_yield_load()` must fill a branch BD-slot." "%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — and the next `atom_label()` token is `%s` — `mac_yield_load()` must fill a branch BD-slot or sit between two `atom_label`s for the natural fall-through load."
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident), , atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
} }
end end
end end
end end
end
-- ── Rule 2: every `mac_yield_tail()` must be at a labeled target whose branch BD-slot is `mac_yield_load()`. -- ── Rule 2: every `mac_yield_tail()` must be at a labeled target whose branch BD-slot is `mac_yield_load()`.
for tok_idx = 1, n do for tok_idx = 1, n do
@@ -2015,8 +2067,9 @@ local function analyze_atom_paths(atom, pipe_ctx)
succ[#succ + 1] = label_pos + 1 succ[#succ + 1] = label_pos + 1
end end
end end
-- For literal-offset jumps (label == false), the target is a non-tracked address; conservatively omit. -- For literal-offset jumps (label == false), control transfers out unconditionally.
return succ, nil -- Treat as a terminator so the path is recorded (NOT as a silent fall-through to the next token, which is unreachable in this atom's execution).
return {}, tok_idx
end end
-- Conditional branch: BD slot absorbed; two successors — fall-through (tok_idx+2) + taken (if known). -- Conditional branch: BD slot absorbed; two successors — fall-through (tok_idx+2) + taken (if known).
if tok_idx + 2 <= n then if tok_idx + 2 <= n then
@@ -2032,9 +2085,11 @@ local function analyze_atom_paths(atom, pipe_ctx)
-- Return (succ, nil), the second value is the terminator marker (nil = not a terminator). -- Return (succ, nil), the second value is the terminator marker (nil = not a terminator).
return succ, nil return succ, nil
end end
-- Normal token: just the next one -- Normal token: just the next one.
-- The final ordinary word of the body has no successor and terminates the path;
-- record it as an implicit endpoint so the cycle budget for non-yield components is not silently zeroed.
if tok_idx + 1 <= n then return { tok_idx + 1 }, nil end if tok_idx + 1 <= n then return { tok_idx + 1 }, nil end
return {}, nil return {}, tok_idx
end end
-- DFS through all paths. Track the current cycle sum, a visited set scoped to the current path (to detect loops), and a count of paths. -- DFS through all paths. Track the current cycle sum, a visited set scoped to the current path (to detect loops), and a count of paths.
+5
View File
@@ -14,16 +14,21 @@ $url_armips = 'https://github.com/Kingcom/armips.git'
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git' $url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git' $url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git' $url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
$path_armips = join-path $path_toolchain 'armips' $path_armips = join-path $path_toolchain 'armips'
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux' $path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu' $path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
$path_lpeg = join-path $path_toolchain 'lpeg' $path_lpeg = join-path $path_toolchain 'lpeg'
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
clone-gitrepo $path_armips $url_armips clone-gitrepo $path_armips $url_armips
clone-gitrepo $path_lpeg $url_lpeg clone-gitrepo $path_lpeg $url_lpeg
clone-gitrepo $path_pcsx_redux $url_pcsx_redux clone-gitrepo $path_pcsx_redux $url_pcsx_redux
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
$path_armips_build = join-path $path_armips 'build' $path_armips_build = join-path $path_armips 'build'
verify-path $path_armips_build verify-path $path_armips_build