mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-08-14 11:38:14 +00:00
Compare commits
15
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b699b47b28 | ||
|
|
640dab7e61 | ||
|
|
4688566767 | ||
|
|
5ebaa6e083 | ||
|
|
7f0bdefbcb | ||
|
|
d5f28b83ea | ||
|
|
3ea3e8d105 | ||
|
|
6b60cef2e8 | ||
|
|
77f19321cd | ||
|
|
2e07665920 | ||
|
|
9501bbbcc2 | ||
|
|
9b6b5535f5 | ||
|
|
7807047dc0 | ||
|
|
7daeec0ee3 | ||
|
|
3f3b691ac0 |
+2
-2
@@ -91,8 +91,8 @@
|
|||||||
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
||||||
#define TSet_(type) type; typedef PtrSet_(type)
|
#define TSet_(type) type; typedef PtrSet_(type)
|
||||||
|
|
||||||
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||||
#define array_decl(type, ...) (type[]){__VA_ARGS__}
|
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||||
#define Array_sym(type,len) A ## len ## _ ## type
|
#define Array_sym(type,len) A ## len ## _ ## type
|
||||||
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
||||||
#define Array_(type,len) Array_expand(type,len)
|
#define Array_(type,len) Array_expand(type,len)
|
||||||
|
|||||||
+56
-7
@@ -60,8 +60,8 @@ WORD_COUNT(mac_yield_tail, 3)
|
|||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||||
load_half( rs_x, r_base, O_(V3_S2,x)) \
|
load_half( rs_x, r_base, offset + O_(V3_S2,x)) \
|
||||||
, load_half( rs_y, r_base, O_(V3_S2,y))
|
, load_half( rs_y, r_base, offset + O_(V3_S2,y))
|
||||||
WORD_COUNT(mac_load_v2s2, 2)
|
WORD_COUNT(mac_load_v2s2, 2)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
@@ -72,9 +72,9 @@ WORD_COUNT(mac_store_v2s2, 2)
|
|||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
|
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
|
||||||
load_word( rs_x, r_base, O_(V3_S4,x)) \
|
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \
|
||||||
, load_word( rs_y, r_base, O_(V3_S4,y)) \
|
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \
|
||||||
, load_word( rs_z, r_base, O_(V3_S4,z))
|
, load_word( rs_z, r_base, offset + O_(V3_S4,z))
|
||||||
WORD_COUNT(mac_load_v3s4, 3)
|
WORD_COUNT(mac_load_v3s4, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
@@ -175,9 +175,58 @@ WORD_COUNT(mac_gte_sqr_v3, 8)
|
|||||||
, shift_aright_var(r_dz, r_dz, r_shift)
|
, shift_aright_var(r_dz, r_dz, r_shift)
|
||||||
WORD_COUNT(mac_gte_gpf_scale, 13)
|
WORD_COUNT(mac_gte_gpf_scale, 13)
|
||||||
|
|
||||||
|
#define mac_apply_matrix_lv(r_mtx, r_vec, r_out, r_t0, r_t1, r_t2) \
|
||||||
|
load_word(r_t0, r_mtx, 0) \
|
||||||
|
, nop \
|
||||||
|
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT11_Code) \
|
||||||
|
, load_word(r_t0, r_mtx, 4) \
|
||||||
|
, nop \
|
||||||
|
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT12_Code) \
|
||||||
|
, load_word(r_t0, r_mtx, 8) \
|
||||||
|
, nop \
|
||||||
|
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT13_Code) \
|
||||||
|
, load_word(r_t0, r_mtx, 12) \
|
||||||
|
, nop \
|
||||||
|
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT21_Code) \
|
||||||
|
, load_half_u(r_t0, r_mtx, 16) \
|
||||||
|
, nop \
|
||||||
|
, gte_mv_to_ctrl_r(r_t0, gte_cr_RT22_Code) \
|
||||||
|
, nop2 /* Load PACKED pos into V0 (libgte SVECTOR layout).
|
||||||
|
* r_vec points to atom-0-staged packed data ((pos.y << 16) | pos.x at +0, pos.z at +4).
|
||||||
|
* LWC2 base register MUST be the pointer r_vec, NOT the loaded value r_t0. */ \
|
||||||
|
, load_word(r_t0, r_vec, 0) \
|
||||||
|
, nop \
|
||||||
|
, gte_lw(C2_VXY0, r_vec, 0) \
|
||||||
|
, load_word(r_t0, r_vec, 4) \
|
||||||
|
, nop \
|
||||||
|
, gte_lw(C2_VZ0, r_vec, 4) /* RTPS: cv=3 (no translation), sf=1 (no shift, integer), v=0 (V0 input),
|
||||||
|
* mx=0 (rotation matrix). MAC = RT row · V0 + 0. RTPS also writes
|
||||||
|
* SXY0/1/2 + SZ0..SZ3 (perspective division); ignored. */ \
|
||||||
|
, gte_cmdw_rtps_sf1 /* Read MAC1/2/3 → out. */ \
|
||||||
|
, gte_mv_from_data_r(r_t0, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(r_t1, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(r_t2, C2_MAC3) \
|
||||||
|
, nop \
|
||||||
|
, store_word(r_t0, r_out, 0) \
|
||||||
|
, store_word(r_t1, r_out, 4) \
|
||||||
|
, store_word(r_t2, r_out, 8)
|
||||||
|
WORD_COUNT(mac_apply_matrix_lv, 31)
|
||||||
|
|
||||||
|
#define mac_trans_matrix(r_mtx, r_off, r_t1) \
|
||||||
|
load_word(r_t1, r_off, O_(V3_S4,x)) \
|
||||||
|
, nop \
|
||||||
|
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[0])) \
|
||||||
|
, load_word(r_t1, r_off, O_(V3_S4,y)) \
|
||||||
|
, nop \
|
||||||
|
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
|
||||||
|
, load_word(r_t1, r_off, O_(V3_S4,z)) \
|
||||||
|
, nop \
|
||||||
|
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[2]))
|
||||||
|
WORD_COUNT(mac_trans_matrix, 9)
|
||||||
|
|
||||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||||
load_upper_i(reg_transfer, cmd >> 16) \
|
load_upper_i(reg_transfer, u4_hi(cmd)) \
|
||||||
, or_i_self( reg_transfer, cmd & 0xFFFF) \
|
, or_i_self( reg_transfer, u4_lo(cmd)) /* load_upper_i(reg_transfer, cmd >> 16), // or_i_self( reg_transfer, cmd & 0xFFFF), */ \
|
||||||
, store_word( reg_transfer, reg_base, port)
|
, store_word( reg_transfer, reg_base, port)
|
||||||
WORD_COUNT(mac_gcmd_push, 3)
|
WORD_COUNT(mac_gcmd_push, 3)
|
||||||
|
|
||||||
|
|||||||
@@ -25,14 +25,14 @@
|
|||||||
#pragma region duffle
|
#pragma region duffle
|
||||||
|
|
||||||
|
|
||||||
// --- atom: normalize_v3s4 (62 words) ---
|
// --- atom: normalize_v3s4 (66 words) ---
|
||||||
|
|
||||||
#define _atom_offset_srav_path_aligned_done 6
|
#define _atom_offset_aligned_done_srav_path 3
|
||||||
#define _atom_offset_aligned_done_srav_path 1
|
#define _atom_offset_srav_path_aligned_done 4
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
|
||||||
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
||||||
|
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- atom: pad_bios_snapshot (84 words) ---
|
// --- atom: pad_bios_snapshot (84 words) ---
|
||||||
|
|||||||
+10
-8
@@ -8,30 +8,32 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom Components)
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gcmd_push(MipsAtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ac_gcmd_push, ab, {
|
MipsAtomComp_Proc_(ac_gcmd_push, ab, {
|
||||||
load_upper_i(reg_transfer, cmd >> 16),
|
load_upper_i(reg_transfer, u4_hi(cmd)),
|
||||||
or_i_self( reg_transfer, cmd & 0xFFFF),
|
or_i_self( reg_transfer, u4_lo(cmd)),
|
||||||
|
// load_upper_i(reg_transfer, cmd >> 16),
|
||||||
|
// or_i_self( reg_transfer, cmd & 0xFFFF),
|
||||||
store_word( reg_transfer, reg_base, port),
|
store_word( reg_transfer, reg_base, port),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rgb8(MipsAtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, {
|
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, {
|
||||||
store_byte(rr, base, offset + O_(RGB8,r)),
|
store_byte(rr, base, offset + O_(RGB8,r)),
|
||||||
store_byte(rg, base, offset + O_(RGB8,g)),
|
store_byte(rg, base, offset + O_(RGB8,g)),
|
||||||
store_byte(rb, base, offset + O_(RGB8,b)),
|
store_byte(rb, base, offset + O_(RGB8,b)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pack_color_word(MipsAtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, {
|
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, {
|
||||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||||
store_word( R_AT, r_base, (off)),
|
store_word( R_AT, r_base, (off)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_format_f3_color(MipsAtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_format_g4_color(MipsAtomBuilder_R ab, U4 r_prim_cursor,
|
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
|
||||||
U1 r0, U1 g0, U1 b0,
|
U1 r0, U1 g0, U1 b0,
|
||||||
U1 r1, U1 g1, U1 b1,
|
U1 r1, U1 g1, U1 b1,
|
||||||
U1 r2, U1 g2, U1 b2,
|
U1 r2, U1 g2, U1 b2,
|
||||||
@@ -44,7 +46,7 @@ MipsAtomComp_Proc_(ac_format_g4_color, ab, {
|
|||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
||||||
I_ Slice_MipsCode ac_insert_ot_tag(MipsAtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, {
|
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, {
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||||
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||||
|
|||||||
+129
-38
@@ -11,7 +11,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
|||||||
#pragma region MACs (Mips Atom Components)
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||||
FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, {
|
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, {
|
||||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
||||||
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
||||||
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||||
@@ -19,14 +19,14 @@ FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4
|
|||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||||
FI_ Slice_MipsCode ac_gte_store_f3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, {
|
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, {
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
||||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
||||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||||
I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, {
|
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, {
|
||||||
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||||
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||||
@@ -37,7 +37,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4
|
|||||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||||
FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, {
|
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, {
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||||
@@ -47,13 +47,13 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cur
|
|||||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||||
*/
|
*/
|
||||||
FI_ Slice_MipsCode ac_gte_store_g4_p3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||||
|
|
||||||
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
||||||
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
|
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
|
||||||
* Stage 2 of normalize consumes these directly.
|
* Stage 2 of normalize consumes these directly.
|
||||||
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
|
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
|
||||||
FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, {
|
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, {
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
@@ -68,7 +68,7 @@ FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz
|
|||||||
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
||||||
* Used standalone for "scale vector by scalar".
|
* Used standalone for "scale vector by scalar".
|
||||||
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
||||||
FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, {
|
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, {
|
||||||
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
@@ -83,6 +83,86 @@ FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r
|
|||||||
shift_aright_var(r_dz, r_dz, r_shift),
|
shift_aright_var(r_dz, r_dz, r_shift),
|
||||||
})
|
})
|
||||||
|
|
||||||
|
/* ─── APPLY MATRIX LV (libgte ApplyMatrixLV port) ───
|
||||||
|
* Atom component — auto-generates mac_apply_matrix_lv Mac composer macro.
|
||||||
|
* Uses GTE RTPS (cv=1, sf=1, v=0) with lwc2-loaded V0/VZ0 inputs.
|
||||||
|
* Per PSX-SPX `geometrytransformationenginegte.md` lines 416-418:
|
||||||
|
* IR1 = MAC1 = (TRX*1000h + RT11*VX0 + RT12*VY0 + RT13*VZ0) SAR (sf*12)
|
||||||
|
* IR2 = MAC2 = (TRY*1000h + RT21*VX0 + RT22*VY0 + RT23*VZ0) SAR (sf*12)
|
||||||
|
* IR3 = MAC3 = (TRZ*1000h + RT31*VX0 + RT32*VY0 + RT33*VZ0) SAR (sf*12)
|
||||||
|
* RTPS uses the FULL row of the rotation matrix (not just diagonal like MVMVA with mx=0).
|
||||||
|
* libgte's `gte_ApplyMatrix` calls `gte_rtv0()` = RTPS cv=1 v=0 mx=0.
|
||||||
|
* Per `gte.h` line 405 the body sets cv=3 (BK, zero-initialized) so no TR contribution.
|
||||||
|
*
|
||||||
|
* Operands:
|
||||||
|
* r_mtx : MT3_S2S4* (matrix pointer)
|
||||||
|
* r_vec : U4 (pointer to PACKED V0 data — (pos.y << 16) | pos.x at +0, pos.z at +4)
|
||||||
|
* r_out : V3_S4* (output pointer; MAC1/2/3 stored here)
|
||||||
|
* r_t0/1/2 : 3 GPR codes for matrix load + intermediate state
|
||||||
|
* Words: ~26. Clobbers: r_t0, r_t1, r_t2 (C2 $0..$4, VXY0/VZ0, MAC1/2/3, SXY0/1/2). */
|
||||||
|
FI_ Slice_MipsCode ac_apply_matrix_lv(AtomBuilder_R ab
|
||||||
|
, U4 r_mtx, U4 r_vec, U4 r_out
|
||||||
|
, U4 r_t0, U4 r_t1, U4 r_t2
|
||||||
|
) MipsAtomComp_Proc_(ac_apply_matrix_lv, ab, {
|
||||||
|
/* Load MATRIX rows into GTE RT11..RT33 (libgte convention: ctc2 to C2 $0..$4 in order).
|
||||||
|
* load_half_u zero-extends the last word so RT33 = m[2][2] and TRX = 0. */
|
||||||
|
load_word(r_t0, r_mtx, 0), nop,
|
||||||
|
gte_mv_to_ctrl_r(r_t0, gte_cr_RT11_Code),
|
||||||
|
load_word(r_t0, r_mtx, 4), nop,
|
||||||
|
gte_mv_to_ctrl_r(r_t0, gte_cr_RT12_Code),
|
||||||
|
load_word(r_t0, r_mtx, 8), nop,
|
||||||
|
gte_mv_to_ctrl_r(r_t0, gte_cr_RT13_Code),
|
||||||
|
load_word(r_t0, r_mtx, 12), nop,
|
||||||
|
gte_mv_to_ctrl_r(r_t0, gte_cr_RT21_Code),
|
||||||
|
load_half_u(r_t0, r_mtx, 16), nop,
|
||||||
|
gte_mv_to_ctrl_r(r_t0, gte_cr_RT22_Code),
|
||||||
|
nop2,
|
||||||
|
|
||||||
|
/* Load PACKED pos into V0 (libgte SVECTOR layout).
|
||||||
|
* r_vec points to atom-0-staged packed data ((pos.y << 16) | pos.x at +0, pos.z at +4).
|
||||||
|
* LWC2 base register MUST be the pointer r_vec, NOT the loaded value r_t0. */
|
||||||
|
load_word(r_t0, r_vec, 0), nop,
|
||||||
|
gte_lw(C2_VXY0, r_vec, 0),
|
||||||
|
load_word(r_t0, r_vec, 4), nop,
|
||||||
|
gte_lw(C2_VZ0, r_vec, 4),
|
||||||
|
|
||||||
|
/* RTPS: cv=3 (no translation), sf=1 (no shift, integer), v=0 (V0 input),
|
||||||
|
* mx=0 (rotation matrix). MAC = RT row · V0 + 0. RTPS also writes
|
||||||
|
* SXY0/1/2 + SZ0..SZ3 (perspective division); ignored. */
|
||||||
|
gte_cmdw_rtps_sf1,
|
||||||
|
|
||||||
|
/* Read MAC1/2/3 → out. */
|
||||||
|
gte_mv_from_data_r(r_t0, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_t1, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_t2, C2_MAC3),
|
||||||
|
nop,
|
||||||
|
store_word(r_t0, r_out, 0),
|
||||||
|
store_word(r_t1, r_out, 4),
|
||||||
|
store_word(r_t2, r_out, 8),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
|
||||||
|
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
|
||||||
|
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
|
||||||
|
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
|
||||||
|
* Words: 9. Clobbers: r_t1. */
|
||||||
|
FI_ Slice_MipsCode ac_trans_matrix(AtomBuilder_R ab
|
||||||
|
, U4 r_mtx, U4 r_off
|
||||||
|
, U4 r_t1
|
||||||
|
) MipsAtomComp_Proc_(ac_trans_matrix, ab, {
|
||||||
|
load_word(r_t1, r_off, O_(V3_S4,x)),
|
||||||
|
nop,
|
||||||
|
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[0])),
|
||||||
|
|
||||||
|
load_word(r_t1, r_off, O_(V3_S4,y)),
|
||||||
|
nop,
|
||||||
|
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
|
||||||
|
|
||||||
|
load_word(r_t1, r_off, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[2])),
|
||||||
|
})
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
#pragma region Atom Procs
|
#pragma region Atom Procs
|
||||||
@@ -126,7 +206,7 @@ FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r
|
|||||||
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||||
*
|
*
|
||||||
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
|
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
|
||||||
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
internal RO_ S2 gte_normalize_sqr_tbl[192] align_(2) = {
|
||||||
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||||
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||||
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
||||||
@@ -165,13 +245,13 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
|||||||
*
|
*
|
||||||
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
|
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
|
||||||
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
|
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
|
||||||
* r_tmp : scratch (reserved for misc use)
|
* r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4
|
||||||
* r_mac1_scratch : MAC1 result scratch (before sum into r_recip_est)
|
* r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3)
|
||||||
* r_mac2_scratch : MAC2 result scratch (clobbered to IR1 in stage 4)
|
* r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra
|
||||||
* r_recip_est : |v|² sum + shift-input + sqrtbl[index] (the main chain)
|
* r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y
|
||||||
* r_lzcr : LZCR value (consumed by stage 3 alignment calc)
|
* r_lzcr : |v|² sum (stage 2) → shift count (stage 3) → 1/|v| (stage 4 IR0)
|
||||||
* r_shift : final srav amount (consumed by stage 4 shift_aright_var)
|
* r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav
|
||||||
* r_branch_tmp : scratch (shift count, branch target, sqrtbl base addr)
|
* r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr)
|
||||||
*
|
*
|
||||||
* Atom_labels are srav_path / aligned_done
|
* Atom_labels are srav_path / aligned_done
|
||||||
* (NOT namespaced — they're internal to this proc;
|
* (NOT namespaced — they're internal to this proc;
|
||||||
@@ -185,7 +265,7 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
|||||||
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
|
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
|
||||||
*/
|
*/
|
||||||
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
|
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
|
||||||
I_ void normalize_v3s4_proc(MipsAtomBuilder_R ab, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */
|
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */
|
||||||
, U4 r_src_offset, U4 r_dst_offset /* GPR codes: PARAMETERIZED offsets (caller passes O_ macros) */
|
, U4 r_src_offset, U4 r_dst_offset /* GPR codes: PARAMETERIZED offsets (caller passes O_ macros) */
|
||||||
, U4 r_src_ptr, U4 r_dst_ptr, U4 r_tmp /* GPR codes: 3 scratch regs (src/dst computed + tmp) */
|
, U4 r_src_ptr, U4 r_dst_ptr, U4 r_tmp /* GPR codes: 3 scratch regs (src/dst computed + tmp) */
|
||||||
, U4 r_mac1_scratch, U4 r_mac2_scratch /* GPR codes: 2 more: MAC1/MAC2 scratch */
|
, U4 r_mac1_scratch, U4 r_mac2_scratch /* GPR codes: 2 more: MAC1/MAC2 scratch */
|
||||||
@@ -193,21 +273,22 @@ I_ void normalize_v3s4_proc(MipsAtomBuilder_R ab, U4 r_scratch /* GPR code: scra
|
|||||||
, U4 r_lzcr, U4 r_shift /* GPR codes: lzcr + final srav amount */
|
, U4 r_lzcr, U4 r_shift /* GPR codes: lzcr + final srav amount */
|
||||||
, U4 r_branch_tmp /* GPR code: scratch (shift count, branch target, lookup addr) */
|
, U4 r_branch_tmp /* GPR code: scratch (shift count, branch target, lookup addr) */
|
||||||
)
|
)
|
||||||
MipsAtom_Proc_(normalize_v3s4, ab, {
|
MipsAtom_Proc_(normalize_v3s4, aa, {
|
||||||
add_si(r_src_ptr, r_scratch, r_src_offset), /* r_src_ptr = &src */
|
add_si(r_src_ptr, r_scratch, r_src_offset), /* r_src_ptr = &src */
|
||||||
add_si(r_dst_ptr, r_scratch, r_dst_offset), /* r_dst_ptr = &dst */
|
add_si(r_dst_ptr, r_scratch, r_dst_offset), /* r_dst_ptr = &dst */
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_mac2_scratch/r_recip_est/r_branch_tmp. */
|
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
|
||||||
load_word(r_mac2_scratch, r_src_ptr, O_(V3_S4,x)),
|
* r_tmp holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */
|
||||||
|
load_word(r_tmp, r_src_ptr, O_(V3_S4,x)),
|
||||||
load_word(r_recip_est, r_src_ptr, O_(V3_S4,y)),
|
load_word(r_recip_est, r_src_ptr, O_(V3_S4,y)),
|
||||||
load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)),
|
load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)),
|
||||||
nop, /* load-delay */
|
nop, /* load-delay */
|
||||||
|
|
||||||
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
||||||
gte_mv_to_data_r(r_mac2_scratch, C2_IR1),
|
gte_mv_to_data_r(r_tmp, C2_IR1),
|
||||||
gte_mv_to_data_r(r_recip_est, C2_IR2),
|
gte_mv_to_data_r(r_recip_est, C2_IR2),
|
||||||
gte_mv_to_data_r(r_branch_tmp, C2_IR3),
|
gte_mv_to_data_r(r_branch_tmp, C2_IR3),
|
||||||
nop, gte_cmdw_sqr,
|
nop, gte_cmdw_sqr,
|
||||||
|
|
||||||
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
|
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
|
||||||
@@ -222,41 +303,51 @@ MipsAtom_Proc_(normalize_v3s4, ab, {
|
|||||||
gte_mv_from_data_r(r_shift, C2_LZCR),
|
gte_mv_from_data_r(r_shift, C2_LZCR),
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* Stage 3: compute srav amount (r_lzcr) + align |v|² to bit 24. */
|
/* Stage 3: compute srav amount (r_lzcr) + align |v|² to bit 24.
|
||||||
|
* IMPORTANT: the sllv/srav below writes the aligned |v|² to r_mac1_scratch (NOT r_lzcr),
|
||||||
|
* so r_lzcr retains the shift count all the way to the start of stage 4.
|
||||||
|
*/
|
||||||
and_i( r_shift, r_shift, -2),
|
and_i( r_shift, r_shift, -2),
|
||||||
|
or_u(r_mac1_scratch, r_lzcr, 0), /* FIX B: save sum before clobbering r_lzcr with shift count */
|
||||||
li_s( r_lzcr, 31),
|
li_s( r_lzcr, 31),
|
||||||
sub_s( r_lzcr, r_lzcr, r_shift),
|
sub_s( r_lzcr, r_lzcr, r_shift),
|
||||||
shift_aright(r_lzcr, r_lzcr, 1),
|
shift_aright(r_lzcr, r_lzcr, 1),
|
||||||
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
|
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
|
||||||
add_si( r_branch_tmp, r_shift, -24),
|
add_si( r_branch_tmp, r_shift, -24),
|
||||||
branch_lt_zero(r_branch_tmp, atom_offset(srav_path, aligned_done)), nop,
|
branch_lt_zero(r_branch_tmp, atom_offset(aligned_done, srav_path)), nop, /* FIX A: bltz → srav_path (LZCR<24 path) */
|
||||||
jump_rel(atom_offset(aligned_done, srav_path)),
|
jump_rel(atom_offset(srav_path, aligned_done)), /* FIX A: b → aligned_done (LZCR>=24 path) */
|
||||||
shift_lleft_var(r_lzcr, r_lzcr, r_branch_tmp), /* when r_branch_tmp < 0 (LZCR < 24): shift r_lzcr left by (24-LZCR) */
|
shift_lleft_var(r_mac1_scratch, r_mac1_scratch, r_branch_tmp), /* FIX B: src=sum (r_mac1_scratch), dst=same */
|
||||||
atom_label(srav_path)
|
atom_label(srav_path)
|
||||||
li_s( r_branch_tmp, 24),
|
li_s( r_branch_tmp, 24),
|
||||||
sub_s( r_branch_tmp, r_branch_tmp, r_shift),
|
sub_s( r_branch_tmp, r_branch_tmp, r_shift),
|
||||||
shift_aright_var(r_lzcr, r_lzcr, r_branch_tmp), /* when r_branch_tmp >= 0 (LZCR >= 24): shift r_lzcr right by (LZCR-24) */
|
shift_aright_var(r_mac1_scratch, r_mac1_scratch, r_branch_tmp), /* FIX B: src=sum (r_mac1_scratch), dst=same */
|
||||||
atom_label(aligned_done)
|
atom_label(aligned_done)
|
||||||
/* r_lzcr holds |v|² aligned to bit 24. */
|
/* Save the shift count to r_shift before the next 5 instructions overwrite r_lzcr
|
||||||
add_si( r_lzcr, r_lzcr, -64),
|
* (the sqrtbl lookup loads 1/|v| into r_lzcr, which becomes IR0 in stage 4). */
|
||||||
shift_lleft(r_lzcr, r_lzcr, 1),
|
or_u(r_shift, r_lzcr, 0), /* r_shift ← shift count (preserved through stage 4) */
|
||||||
|
/* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */
|
||||||
|
add_si( r_mac1_scratch, r_mac1_scratch, -64),
|
||||||
|
shift_lleft(r_mac1_scratch, r_mac1_scratch, 1),
|
||||||
load_upper_i(r_branch_tmp, u4_hi(& gte_normalize_sqr_tbl)),
|
load_upper_i(r_branch_tmp, u4_hi(& gte_normalize_sqr_tbl)),
|
||||||
or_i_self( r_branch_tmp, u4_lo(& gte_normalize_sqr_tbl)),
|
or_i_self( r_branch_tmp, u4_lo(& gte_normalize_sqr_tbl)),
|
||||||
add_u(r_branch_tmp, r_branch_tmp, r_lzcr),
|
add_u(r_branch_tmp, r_branch_tmp, r_mac1_scratch),
|
||||||
load_half(r_lzcr, r_branch_tmp, 0), nop,
|
load_half(r_lzcr, r_branch_tmp, 0), nop, /* r_lzcr = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
|
||||||
|
|
||||||
/* Stage 4: GPF + srav finalize (r_lzcr = srav_amount carried from stage 3). */
|
/* FIX bug C: r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */
|
||||||
|
load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)), nop, /* r_branch_tmp = src.z (for IR3 in stage 4) */
|
||||||
|
|
||||||
|
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_lzcr = 1/|v|). */
|
||||||
gte_mv_to_data_r(r_lzcr, C2_IR0),
|
gte_mv_to_data_r(r_lzcr, C2_IR0),
|
||||||
gte_mv_to_data_r(r_mac2_scratch, C2_IR1),
|
gte_mv_to_data_r(r_tmp, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||||
gte_mv_to_data_r(r_recip_est, C2_IR2),
|
gte_mv_to_data_r(r_recip_est, C2_IR2),
|
||||||
gte_mv_to_data_r(r_branch_tmp, C2_IR3),
|
gte_mv_to_data_r(r_branch_tmp, C2_IR3), /* IR3 = src.z (reloaded) */
|
||||||
nop2, gte_cmdw_gpf,
|
nop2, gte_cmdw_gpf,
|
||||||
gte_mv_from_data_r(r_mac2_scratch, C2_MAC1),
|
gte_mv_from_data_r(r_mac2_scratch, C2_MAC1),
|
||||||
gte_mv_from_data_r(r_recip_est, C2_MAC2),
|
gte_mv_from_data_r(r_recip_est, C2_MAC2),
|
||||||
gte_mv_from_data_r(r_branch_tmp, C2_MAC3),
|
gte_mv_from_data_r(r_branch_tmp, C2_MAC3),
|
||||||
shift_aright_var(r_mac2_scratch, r_mac2_scratch, r_lzcr),
|
shift_aright_var(r_mac2_scratch, r_mac2_scratch, r_shift), /* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
|
||||||
shift_aright_var(r_recip_est, r_recip_est, r_lzcr),
|
shift_aright_var(r_recip_est, r_recip_est, r_shift),
|
||||||
shift_aright_var(r_branch_tmp, r_branch_tmp, r_lzcr),
|
shift_aright_var(r_branch_tmp, r_branch_tmp, r_shift),
|
||||||
|
|
||||||
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
|
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
|
||||||
store_word(r_mac2_scratch, r_dst_ptr, O_(V3_S4,x)),
|
store_word(r_mac2_scratch, r_dst_ptr, O_(V3_S4,x)),
|
||||||
|
|||||||
@@ -191,6 +191,30 @@ enum {
|
|||||||
gte_mask_fake_cmd = 0x1F,
|
gte_mask_fake_cmd = 0x1F,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
||||||
|
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon:
|
||||||
|
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
||||||
|
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
||||||
|
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
||||||
|
* Cross-alias writes inside one atom body, or across the wave-context boundary,
|
||||||
|
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes
|
||||||
|
* (CHECK_RULES row) warns about each pair per source. See
|
||||||
|
* docs/gte_reference.md §"Control-register alias table" for the silicon
|
||||||
|
* rationale and the libgte outer-product convention.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
||||||
|
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
||||||
|
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
||||||
|
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
||||||
|
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product
|
||||||
|
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then
|
||||||
|
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent
|
||||||
|
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
|
||||||
|
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
|
||||||
|
* for the canonical write pattern.
|
||||||
|
*/
|
||||||
|
|
||||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||||
@@ -391,6 +415,56 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
||||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
||||||
|
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
|
||||||
|
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
|
||||||
|
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
|
||||||
|
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
|
||||||
|
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
|
||||||
|
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
|
||||||
|
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
|
||||||
|
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
|
||||||
|
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
|
||||||
|
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
|
||||||
|
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
|
||||||
|
* The combine is (pass1 << 3) + pass2. */
|
||||||
|
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
|
||||||
|
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
|
||||||
|
|
||||||
|
/* MVMVA pass 2 — EXACT C11 ApplyMatrixLV command.
|
||||||
|
* Command word: 0x4A49E012.
|
||||||
|
* bits 31-26: 010010 = COP2
|
||||||
|
* bit 25: 1 (CO set)
|
||||||
|
* bits 24-20: 01001 = 9 (fake_cmd)
|
||||||
|
* bit 19: 1 (sf=1)
|
||||||
|
* bits 18-17: 00 (mx=0, RT matrix)
|
||||||
|
* bits 16-15: 11 (v=3, IR)
|
||||||
|
* bits 14-13: 11 (cv=3, no translation)
|
||||||
|
* bits 5-0: 010010 = MVMVA
|
||||||
|
* sf=1, mx=0, v=3, cv=3. Pass 2 reads RT matrix, IR input, >>12. */
|
||||||
|
#define gte_cmdw_mvmva_c11_pass2_exact 0x4A49E012
|
||||||
|
|
||||||
|
/* MVMVA pass 1 — C11's exact command: 0x4A41E012.
|
||||||
|
* bit 25: 1, sf=0, mx=0, v=3, cv=3. Pass 1 reads RT matrix, IR input, no shift. */
|
||||||
|
#define gte_cmdw_mvmva_c11_pass1_exact 0x4A41E012
|
||||||
|
|
||||||
|
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
||||||
|
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's
|
||||||
|
* ApplyMatrixLV when the GTE pipeline expects R*pos >> 12. The shift produces
|
||||||
|
* values like (-270, 710, 1713) which match the C11 reference path. */
|
||||||
|
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
||||||
|
|
||||||
/* SQR / GPF cosmetic-bits compat helpers.
|
/* SQR / GPF cosmetic-bits compat helpers.
|
||||||
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
||||||
* The hardware ignores these bits (per PSX-SPX line 48). */
|
* The hardware ignores these bits (per PSX-SPX line 48). */
|
||||||
|
|||||||
+79
-34
@@ -23,11 +23,11 @@
|
|||||||
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||||
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
||||||
*
|
*
|
||||||
* This behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
||||||
* to author and compose programs with. From here various conventions can be further applied.
|
* to author and compose programs with. From here various conventions can be further applied.
|
||||||
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||||
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
||||||
* Branching nearly is always downstream. Atuomatic stack usage is non-existent.
|
* Branching nearly is always downstream. Automatic stack usage is non-existent.
|
||||||
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
||||||
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
||||||
*
|
*
|
||||||
@@ -39,10 +39,10 @@
|
|||||||
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
||||||
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
||||||
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
||||||
* automatic register allocation means the user cannott ignore excessive argument shuffle across workload or
|
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
|
||||||
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
||||||
*
|
*
|
||||||
* Learning data-oreinted code becomes a natural progression. Your not fighting a stack-based procedural
|
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
|
||||||
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
||||||
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
||||||
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
||||||
@@ -100,17 +100,27 @@ enum {
|
|||||||
// S 0-7
|
// S 0-7
|
||||||
};
|
};
|
||||||
|
|
||||||
|
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
||||||
|
|
||||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||||
typedef Slice_(MipsCode);
|
typedef Slice_(MipsCode);
|
||||||
|
|
||||||
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
|
typedef U4 const MipsAtom;
|
||||||
|
typedef Slice_(MipsAtom);
|
||||||
|
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
||||||
|
// MipsAtom* <identifier>[...];
|
||||||
|
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
||||||
|
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the blow returns 'MipsAtom'.
|
||||||
|
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
||||||
|
|
||||||
|
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
||||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
// Used for atoms with value-args
|
// Used for atoms with value-args
|
||||||
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
// FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
||||||
// expands to:
|
// expands to:
|
||||||
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; }
|
// FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; }
|
||||||
#define MipsAtom_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); }
|
#define MipsAtom_Proc_(sym, aa, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, sym)); }
|
||||||
|
|
||||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||||
// MipsAtomComp_(ac_X) { body }
|
// MipsAtomComp_(ac_X) { body }
|
||||||
@@ -128,7 +138,7 @@ typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that
|
|||||||
// The body must NOT include mac_yield() (the parent atom yields).
|
// The body must NOT include mac_yield() (the parent atom yields).
|
||||||
// Inline-only callers (the generated `mac_<name>` aliases) skip this arg via metaprogram filtering;
|
// Inline-only callers (the generated `mac_<name>` aliases) skip this arg via metaprogram filtering;
|
||||||
// escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
// escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
||||||
#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(ab, slice_from_array(MipsCode, sym)); }
|
#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, sym)); }
|
||||||
|
|
||||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||||
Files containing only atoms and atom components.
|
Files containing only atoms and atom components.
|
||||||
@@ -139,10 +149,10 @@ typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that
|
|||||||
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
|
(the identifier embeds the source line, so duplicates across `#include`d files don't collide). */
|
||||||
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||||
|
|
||||||
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
|
typedef Slice_MipsAtom Tape;
|
||||||
|
|
||||||
/* The 'Exit' Atom */
|
/* The 'Exit' Atom */
|
||||||
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(R_RA), nop };
|
||||||
|
|
||||||
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
|
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
|
||||||
|
|
||||||
@@ -231,43 +241,78 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
|||||||
};
|
};
|
||||||
#pragma endregion Macro Atom Components
|
#pragma endregion Macro Atom Components
|
||||||
|
|
||||||
#pragma region Mips Atom Builder
|
#pragma region Atom Builder
|
||||||
// This helps with runtime procedural authoring of mips atoms.
|
// This helps with runtime procedural authoring of mips atoms.
|
||||||
|
|
||||||
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
||||||
|
|
||||||
// FArena Related
|
// FArena Related
|
||||||
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||||
// Whatever the builder is writting to should most likely coresspond
|
|
||||||
// to something that can fit within instruction cache?
|
|
||||||
|
|
||||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode code) {
|
// Usual way to resolve an atom after the bulder is done.
|
||||||
/* code.len is in ELEMENTS (per slice_from_array convention); ab->used is also in elements
|
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
|
||||||
* (the init uses `ab->used * sizeof(U4)` for byte offset arithmetic — sizeof(U4)==4==sizeof(MipsCode)).
|
|
||||||
* mem_copy needs BYTES, so we use S_slice(code) for the length. */
|
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
|
||||||
assert(ab->capacity - ab->used - code.len);
|
assert(ab->capacity - ab->used - code.len);
|
||||||
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot (arena accumulation) */
|
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
|
||||||
mem_copy(u4_(dest), u4_(code.ptr), S_slice(code));
|
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
|
||||||
mem_bump(ab->start, ab->capacity, & ab->used, code.len);
|
|
||||||
}
|
}
|
||||||
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||||
|
|
||||||
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
||||||
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
|
||||||
/* ac_yield is a MipsCode[] of 4 elements; S_(ac_yield)=bytes, array_len(ac_yield)=elements.
|
|
||||||
* ab->used is in elements, so mem_bump needs element count. */
|
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
|
||||||
U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot */
|
#pragma endregion Mips Atom Builder
|
||||||
mem_copy(u4_(dest), u4_(ac_yield), S_(ac_yield));
|
|
||||||
mem_bump(ab->start, ab->capacity, & ab->used, array_len(ac_yield));
|
#pragma region Atom Arena
|
||||||
|
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
||||||
|
|
||||||
|
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
||||||
|
|
||||||
|
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
||||||
|
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
|
arena->start = u4_(mem.ptr);
|
||||||
|
arena->capacity = mem.len;
|
||||||
|
arena->used = 0;
|
||||||
|
}
|
||||||
|
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
|
||||||
|
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
|
||||||
|
assert(aa->capacity - aa->used - code.len);
|
||||||
|
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
|
||||||
|
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
|
||||||
|
return C_(MipsAtom*, dest);
|
||||||
|
}
|
||||||
|
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
||||||
|
#pragma region Atom Arena
|
||||||
|
|
||||||
|
#pragma region RegFile (Register File Allocator)
|
||||||
|
// A specialized allocator utilized to help the user track which registers are bound to values
|
||||||
|
// that must be preserved for the arena's bounds.
|
||||||
|
|
||||||
|
enum {
|
||||||
|
RegFileArena_Len,
|
||||||
|
};
|
||||||
|
typedef Enum_(U4, RegFileEntry) {
|
||||||
|
// TODO(Ed): Define RF_Field, each field is maped by index + bit pos.
|
||||||
|
// the index is the upper portion of a U4 and the bit pos in the lower pos.
|
||||||
|
|
||||||
|
regfileentry_todo_,
|
||||||
|
// TODO(Ed): Is there a trick we can do with the current register enums to
|
||||||
|
// just resolve an entry automatically when doing a pin?
|
||||||
|
};
|
||||||
|
typedef Struct_(RegFile) {
|
||||||
|
U1 GPR[RegFileArena_Len];
|
||||||
|
U1 GTE[RegFileArena_Len];
|
||||||
|
U1 GP[RegFileArena_Len];
|
||||||
|
};
|
||||||
|
|
||||||
|
void regfile_pin(U4 register) {
|
||||||
|
|
||||||
|
assert(false);
|
||||||
}
|
}
|
||||||
|
|
||||||
#define mipsatom_from_builder(ab) C_(MipsAtom*, (ab).start)
|
#pragma endregion RegFileArena (Register File Allocator)
|
||||||
|
|
||||||
// tb_emit_builder(tb, ab) — emit the builder's atom into the tape and advance tb->used.
|
|
||||||
// Thin wrapper around tb_emit(tb, mipsatom_from_builder(ab[0])).
|
|
||||||
// Equivalent to tb_emit(tb, code_<name>) for runtime-built atoms.
|
|
||||||
FI_ void tb_emit_builder(TapeBuilder_R tb, MipsAtomBuilder_R ab) { tb_emit(tb, mipsatom_from_builder(ab[0])); }
|
|
||||||
#pragma endregion Mips Atom Builder
|
|
||||||
|
|
||||||
#pragma region Mips Atom Procs
|
#pragma region Mips Atom Procs
|
||||||
|
|
||||||
|
|||||||
+17
-11
@@ -9,35 +9,41 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom Component)
|
#pragma region MACs (Mips Atom Component)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v2s2(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, {
|
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, {
|
||||||
load_half( rs_x, r_base, O_(V3_S2,x)),
|
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
|
||||||
load_half( rs_y, r_base, O_(V3_S2,y)),
|
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v2s2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, {
|
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, {
|
||||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v3s4(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, {
|
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, {
|
||||||
load_word( rs_x, r_base, O_(V3_S4,x)),
|
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
|
||||||
load_word( rs_y, r_base, O_(V3_S4,y)),
|
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
|
||||||
load_word( rs_z, r_base, O_(V3_S4,z)),
|
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
|
||||||
})
|
})
|
||||||
|
// TODO(Ed): we could generate these mappings properly..
|
||||||
|
#define ac_load_p3s4 ac_load_v3s4
|
||||||
|
#define mac_load_p3s4 mac_load_v3s4
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v3s4(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, {
|
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, {
|
||||||
store_word(rt_x, base, offset + O_(V3_S4,x)),
|
store_word(rt_x, base, offset + O_(V3_S4,x)),
|
||||||
store_word(rt_y, base, offset + O_(V3_S4,y)),
|
store_word(rt_y, base, offset + O_(V3_S4,y)),
|
||||||
store_word(rt_z, base, offset + O_(V3_S4,z)),
|
store_word(rt_z, base, offset + O_(V3_S4,z)),
|
||||||
})
|
})
|
||||||
|
// TODO(Ed): we could generate these mappings properly..
|
||||||
|
#define ac_store_p3s4 ac_store_v3s4
|
||||||
|
#define mac_store_p3s4 mac_store_v3s4
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_sub_v3s4(MipsAtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, {
|
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, {
|
||||||
sub_s(rds_x, rds_x, rt_x),
|
sub_s(rds_x, rds_x, rt_x),
|
||||||
sub_s(rds_y, rds_y, rt_y),
|
sub_s(rds_y, rds_y, rt_y),
|
||||||
sub_s(rds_z, rds_z, rt_z),
|
sub_s(rds_z, rds_z, rt_z),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rects2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, {
|
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, {
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||||
|
|||||||
+16
-7
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
|
|||||||
|
|
||||||
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
||||||
|
|
||||||
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
|
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
|
||||||
assert(amount <= (cap - used[0]));
|
assert(amount <= (cap - used[0]));
|
||||||
used[0] += amount;
|
used[0] += amount;
|
||||||
}
|
}
|
||||||
@@ -72,10 +72,10 @@ typedef Slice_(B1);
|
|||||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||||
|
|
||||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
|
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
|
||||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) / S_(type) } /* .len in elements (matches S_slice/slice_arg_from_array convention) */
|
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) / S_(type) }
|
||||||
|
|
||||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), S_slice(s)); }
|
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
|
||||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||||
|
|
||||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||||
@@ -89,6 +89,13 @@ FI_ void slice_copy_(Slice dest, Slice src) {
|
|||||||
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
||||||
} while(0)
|
} while(0)
|
||||||
|
|
||||||
|
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
|
||||||
|
assert(len - used[0] - amount);
|
||||||
|
U4 ptr = start + used[0]; used[0] += amount;
|
||||||
|
return slice_ut(ptr, amount);
|
||||||
|
}
|
||||||
|
|
||||||
|
typedef Slice_(U1);
|
||||||
typedef Slice_(U4);
|
typedef Slice_(U4);
|
||||||
|
|
||||||
#pragma endregion Slice
|
#pragma endregion Slice
|
||||||
@@ -99,16 +106,17 @@ typedef Opt_(farena) { U4 alignment, type_width; };
|
|||||||
typedef Struct_(FArena) { U4 start, capacity, used; };
|
typedef Struct_(FArena) { U4 start, capacity, used; };
|
||||||
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
arena->start = u4_(mem.ptr);
|
arena->start = u4_(mem.ptr);
|
||||||
arena->capacity = S_slice(mem); /* FArena.used is in BYTES; capacity must be bytes too */
|
arena->capacity = mem.len;
|
||||||
arena->used = 0;
|
arena->used = 0;
|
||||||
}
|
}
|
||||||
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
||||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
|
||||||
|
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||||
if (amount == 0) { return (Slice){}; }
|
if (amount == 0) { return (Slice){}; }
|
||||||
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
||||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||||
U4 ptr = arena->start + arena->used;
|
U4 ptr = arena->start + arena->used;
|
||||||
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
|
mem_bump(arena->capacity, & arena->used, to_commit);
|
||||||
return (Slice){ (B1*)ptr, to_commit };
|
return (Slice){ (B1*)ptr, to_commit };
|
||||||
}
|
}
|
||||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||||
@@ -117,6 +125,7 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
|||||||
arena->used -= save_point - arena->start;
|
arena->used -= save_point - arena->start;
|
||||||
}
|
}
|
||||||
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
||||||
|
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
||||||
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
||||||
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
||||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||||
|
|||||||
@@ -20,14 +20,14 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
|||||||
* 6. sp += 8
|
* 6. sp += 8
|
||||||
*/
|
*/
|
||||||
internal MipsAtom_(mips_flush_icache) {
|
internal MipsAtom_(mips_flush_icache) {
|
||||||
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
|
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
|
||||||
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
|
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
|
||||||
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
|
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
|
||||||
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
|
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||||
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
|
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
|
||||||
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
|
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
|
||||||
jump_reg(rret_addr), // jr $ra
|
jump_reg(R_RA), // jr $ra
|
||||||
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
|
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (BD)
|
||||||
mac_yield(),
|
mac_yield(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+26
-26
@@ -136,31 +136,31 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
||||||
|
|
||||||
, rdiscard = R_0 /* Hardwired to 0 */
|
// , rdiscard = R_0 /* Hardwired to 0 */
|
||||||
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||||
, rret_0 = R_V0 /* Function return value */
|
// , rret_0 = R_V0 /* Function return value */
|
||||||
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||||
, rarg_0 = R_A0 /* First function argument */
|
// , rarg_0 = R_A0 /* First function argument */
|
||||||
, rarg_1 = R_A1 /* Second function argument */
|
// , rarg_1 = R_A1 /* Second function argument */
|
||||||
, rarg_2 = R_A2 /* Third function argument */
|
// , rarg_2 = R_A2 /* Third function argument */
|
||||||
, rarg_3 = R_A3 /* Fourth function argument */
|
// , rarg_3 = R_A3 /* Fourth function argument */
|
||||||
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||||
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||||
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||||
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||||
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||||
, rstatic_1 = R_S1
|
// , rstatic_1 = R_S1
|
||||||
, rstatic_2 = R_S2
|
// , rstatic_2 = R_S2
|
||||||
, rstatic_3 = R_S3
|
// , rstatic_3 = R_S3
|
||||||
, rstatic_4 = R_S4
|
// , rstatic_4 = R_S4
|
||||||
, rstatic_5 = R_S5
|
// , rstatic_5 = R_S5
|
||||||
, rstatic_6 = R_S6
|
// , rstatic_6 = R_S6
|
||||||
, rstatic_7 = R_S7
|
// , rstatic_7 = R_S7
|
||||||
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||||
, rstack_ptr = R_SP /* Stack Pointer */
|
// , rstack_ptr = R_SP /* Stack Pointer */
|
||||||
, rret_addr = R_RA /* Return Address (populated by JAL) */
|
// , rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||||
|
|
||||||
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
||||||
|
|
||||||
@@ -453,7 +453,7 @@ enum { _BitOffsets = 0
|
|||||||
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
||||||
|
|
||||||
/* nop — sll $0, $0, 0 */
|
/* nop — sll $0, $0, 0 */
|
||||||
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
#define nop shift_lleft(R_0, R_0, 0)
|
||||||
#define nop2 nop, nop
|
#define nop2 nop, nop
|
||||||
|
|
||||||
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
||||||
|
|||||||
@@ -11,18 +11,18 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom Components)
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_centered_axes(MipsAtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, {
|
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, {
|
||||||
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF),
|
load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF),
|
||||||
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF),
|
or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF),
|
||||||
store_word( r_scratch, r_state, O_(PadState,axes)),
|
store_word( r_scratch, r_state, O_(PadState,axes)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_id_byte(MipsAtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, {
|
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, {
|
||||||
add_ui( r_id, R_0, id_value),
|
add_ui( r_id, R_0, id_value),
|
||||||
store_byte(r_id, r_state, O_(PadState,id)),
|
store_byte(r_id, r_state, O_(PadState,id)),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, {
|
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, {
|
||||||
add_ui( r_tmp, R_0, pad_status),
|
add_ui( r_tmp, R_0, pad_status),
|
||||||
store_word(r_tmp, r_state, O_(PadState,status)),
|
store_word(r_tmp, r_state, O_(PadState,status)),
|
||||||
})
|
})
|
||||||
@@ -30,7 +30,7 @@ FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state,
|
|||||||
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
||||||
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
||||||
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
||||||
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(MipsAtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, {
|
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, {
|
||||||
nor_u( r_buttons, r_buttons, R_0),
|
nor_u( r_buttons, r_buttons, R_0),
|
||||||
store_half( r_buttons, r_pad_state, O_(PadState, buttons)),
|
store_half( r_buttons, r_pad_state, O_(PadState, buttons)),
|
||||||
})
|
})
|
||||||
|
|||||||
+10
-10
@@ -36,13 +36,13 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
|||||||
* $t2 = 0xB0 (BIOS B-table address) */
|
* $t2 = 0xB0 (BIOS B-table address) */
|
||||||
asm volatile(
|
asm volatile(
|
||||||
asm_words(
|
asm_words(
|
||||||
or_u( rarg_2, rarg_1, rdiscard), /* $a2 = $a1 = raw1 */
|
or_u( R_A2, R_A0, R_0), /* $a2 = $a1 = raw1 */
|
||||||
add_ui( rarg_1, rdiscard, bios_pad_buffer_size), /* $a1 = 0x22 */
|
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
|
||||||
add_ui( rarg_3, rdiscard, bios_pad_buffer_size), /* $a3 = 0x22 */
|
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
|
||||||
add_ui( rtmp_1, rdiscard, bios_init_pad_2), /* $t1 = 0x12 */
|
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
|
||||||
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 */
|
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
|
||||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
call_reg(R_T2), /* jalr $t2, $ra */
|
||||||
nop /* BD slot */
|
nop /* BD slot */
|
||||||
)
|
)
|
||||||
asm_rpins, r_use(p0), r_use(p1)
|
asm_rpins, r_use(p0), r_use(p1)
|
||||||
asm_clobber:
|
asm_clobber:
|
||||||
@@ -62,9 +62,9 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
|||||||
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||||
asm volatile(
|
asm volatile(
|
||||||
asm_words(
|
asm_words(
|
||||||
add_ui( rtmp_1, rdiscard, bios_start_pad_2), /* $t1 = 0x13 */
|
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
|
||||||
add_ui( rtmp_2, rdiscard, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
||||||
call_reg(rtmp_2), /* jalr $t2, $ra */
|
call_reg(R_T2), /* jalr $t2, $ra */
|
||||||
nop /* BD slot */
|
nop /* BD slot */
|
||||||
)
|
)
|
||||||
asm_clobber:
|
asm_clobber:
|
||||||
|
|||||||
@@ -15,6 +15,8 @@
|
|||||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
|
||||||
WORD_COUNT(nop, 1)
|
WORD_COUNT(nop, 1)
|
||||||
|
WORD_COUNT(atom_label, 0)
|
||||||
|
WORD_COUNT(atom_offset, 0)
|
||||||
WORD_COUNT(load_upper_i, 1)
|
WORD_COUNT(load_upper_i, 1)
|
||||||
WORD_COUNT(jump_reg, 1)
|
WORD_COUNT(jump_reg, 1)
|
||||||
WORD_COUNT(jump_link, 1)
|
WORD_COUNT(jump_link, 1)
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
|||||||
|
|
||||||
#pragma region MACs (Mips Atom components)
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
|
MipsAtomComp_Proc_(ac_put_disp_env, ab, {
|
||||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
@@ -36,7 +36,7 @@ MipsAtomComp_Proc_(ac_put_disp_env, ab, {
|
|||||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
|
MipsAtomComp_Proc_(ac_put_draw_env, ab, {
|
||||||
/*
|
/*
|
||||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
@@ -123,6 +123,7 @@ enum {
|
|||||||
/* Wave-context GPR carrier for the resolve_look_at bundle: the scratch base.
|
/* Wave-context GPR carrier for the resolve_look_at bundle: the scratch base.
|
||||||
* Set by atom 0 (popped from tape), read by atoms 1-6 (used as pointer base). */
|
* Set by atom 0 (popped from tape), read by atoms 1-6 (used as pointer base). */
|
||||||
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
|
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
|
||||||
|
#define R_ResolveScratch_Code R_T4_Code
|
||||||
};
|
};
|
||||||
typedef Struct_(Binds_ResolveLookAt) {
|
typedef Struct_(Binds_ResolveLookAt) {
|
||||||
MT3_S2S4* look_at;
|
MT3_S2S4* look_at;
|
||||||
@@ -131,11 +132,6 @@ typedef Struct_(Binds_ResolveLookAt) {
|
|||||||
V3_S4* up_in;
|
V3_S4* up_in;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* Per-atom bind-pop structs for the resolve_look_at bundle. */
|
|
||||||
typedef Struct_(Binds_ResolveLookAtScratch) {
|
|
||||||
U4 scratch_base; /* U4 (scratch base address — populated by helper with u4_(smem.scratchpad)) */
|
|
||||||
};
|
|
||||||
|
|
||||||
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's
|
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's
|
||||||
* scratchpad slots (PS1 hardware scratchpad at 0x1F800000).
|
* scratchpad slots (PS1 hardware scratchpad at 0x1F800000).
|
||||||
*
|
*
|
||||||
@@ -192,27 +188,17 @@ typedef Struct_(ResolveLookAtScratch) {
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
typedef Struct_(Binds_ResolveLookAtSub) {
|
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||||
U4 target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
|
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
|
||||||
U4 eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||||
U4 up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||||
|
ResolveLookAtScratch* scratchpad;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
|
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
|
||||||
* Inputs (C-side pointers popped from the tape):
|
|
||||||
* r_target_ptr : P3_S4* (C-side struct; atom 0 reads target.x/y/z directly)
|
|
||||||
* r_eye_ptr : P3_S4* (C-side struct; staged into scratchpad at +96/+100/+104)
|
|
||||||
* r_up_in_ptr : V3_S4* (C-side struct; staged into scratchpad at +128/+132/+136)
|
|
||||||
* Wave-context output:
|
|
||||||
* r_scratch : R_ResolveScratch (R_T4) — scratch base, read by atoms 1-6
|
|
||||||
*
|
|
||||||
* Bind-pop layout:
|
|
||||||
* Binds_ResolveLookAtSub = 12 bytes (target + eye + up_in ptrs)
|
|
||||||
* Binds_ResolveLookAtScratch = 4 bytes (scratch_base)
|
|
||||||
* Staging work:
|
* Staging work:
|
||||||
* * Stage eye.x/y/z → scratch+96/+100/+104 (for atom 6's translation column)
|
* * Stage eye.x/y/z → scratch (for atom 6's translation column)
|
||||||
* * Stage up_in.x/y/z → scratch+128/+132/+136 (for atom 2's outer-product operand)
|
* * Stage up_in.x/y/z → scratch (for atom 2's outer-product operand)
|
||||||
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
|
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
|
||||||
*
|
|
||||||
* GPR codes (assigned by resolve_look_at_init):
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
* r_target_ptr : R_T0
|
* r_target_ptr : R_T0
|
||||||
* r_eye_ptr : R_T1
|
* r_eye_ptr : R_T1
|
||||||
@@ -224,57 +210,35 @@ typedef Struct_(Binds_ResolveLookAtSub) {
|
|||||||
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
|
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
|
||||||
* R_AT : hardcoded (load eye.y / eye.z / target.z)
|
* R_AT : hardcoded (load eye.y / eye.z / target.z)
|
||||||
* R_V0 : hardcoded (load eye.z / target.z)
|
* R_V0 : hardcoded (load eye.z / target.z)
|
||||||
*
|
|
||||||
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
|
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
|
||||||
*/
|
*/
|
||||||
I_ void resolve_look_at__input_and_sub_proc(MipsAtomBuilder_R ab, U4 r_scratch
|
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa,
|
||||||
|
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
|
||||||
|
U4 r_scratch
|
||||||
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
|
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
|
||||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
|
||||||
) MipsAtom_Proc_(resolve_look_at__input_and_sub, ab, {
|
) MipsAtom_Proc_(resolve_look_at__input_and_sub, aa, {
|
||||||
/* Pop the 3 C-side pointers + scratch_base from the tape. */
|
|
||||||
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||||
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||||
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||||
|
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||||
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtScratch,scratch_base)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtScratch)),
|
|
||||||
|
|
||||||
/* Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation
|
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
||||||
* column). Reuse r_tmp0/r_tmp1/r_tmp2. Offsets via O_(ResolveLookAtScratch,*). */
|
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_eye_ptr, 0),
|
||||||
load_word(r_tmp0, r_eye_ptr, O_(P3_S4,x)),
|
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||||
load_word(r_tmp1, r_eye_ptr, O_(P3_S4,y)),
|
|
||||||
load_word(r_tmp2, r_eye_ptr, O_(P3_S4,z)),
|
|
||||||
nop, /* load-delay */
|
|
||||||
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,eye.x)),
|
|
||||||
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,eye.y)),
|
|
||||||
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye.z)),
|
|
||||||
|
|
||||||
/* Stage up_in.x/y/z into the scratchpad (atom 2 reads these for the outer
|
/* Stage up_in.x/y/z into the scratchpad. */
|
||||||
* product with uz). Reuse r_tmp0/r_tmp1/r_tmp2. */
|
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_up_in_ptr, 0),
|
||||||
load_word(r_tmp0, r_up_in_ptr, O_(V3_S4,x)),
|
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in)),
|
||||||
load_word(r_tmp1, r_up_in_ptr, O_(V3_S4,y)),
|
|
||||||
load_word(r_tmp2, r_up_in_ptr, O_(V3_S4,z)),
|
|
||||||
nop, /* load-delay */
|
|
||||||
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,up_in.x)),
|
|
||||||
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,up_in.y)),
|
|
||||||
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in.z)),
|
|
||||||
|
|
||||||
/* Compute fwd = target - eye. */
|
/* Compute fwd = target - eye. */
|
||||||
load_word(r_tmp0, r_target_ptr, O_(P3_S4,x)),
|
mac_load_p3s4(r_tmp0, r_tmp1, r_tmp2, r_target_ptr, 0),
|
||||||
load_word(r_tmp1, r_target_ptr, O_(P3_S4,y)),
|
mac_load_p3s4(r_tmp3, R_AT, R_V0, r_eye_ptr, 0),
|
||||||
load_word(r_tmp2, r_target_ptr, O_(P3_S4,z)),
|
mac_sub_v3s4(
|
||||||
load_word(r_tmp3, r_eye_ptr, O_(P3_S4,x)),
|
r_tmp0, r_tmp1, r_tmp2,
|
||||||
load_word(R_AT, r_eye_ptr, O_(P3_S4,y)),
|
r_tmp3, R_AT, R_V0),
|
||||||
load_word(R_V0, r_eye_ptr, O_(P3_S4,z)),
|
mac_store_v3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd)),
|
||||||
nop, /* load-delay */
|
|
||||||
sub_u(r_tmp0, r_tmp0, r_tmp3),
|
|
||||||
sub_u(r_tmp1, r_tmp1, R_AT),
|
|
||||||
sub_u(r_tmp2, r_tmp2, R_V0),
|
|
||||||
|
|
||||||
/* Store fwd.x/y/z (atom 1 reads these as the normalize src). */
|
|
||||||
store_word(r_tmp0, r_scratch, O_(ResolveLookAtScratch,fwd.x)),
|
|
||||||
store_word(r_tmp1, r_scratch, O_(ResolveLookAtScratch,fwd.y)),
|
|
||||||
store_word(r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd.z)),
|
|
||||||
|
|
||||||
mac_yield()
|
mac_yield()
|
||||||
})
|
})
|
||||||
@@ -295,12 +259,12 @@ I_ void resolve_look_at__input_and_sub_proc(MipsAtomBuilder_R ab, U4 r_scratch
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
/* Atom 2: cross uz × up_in → right. */
|
/* Atom 2: cross uz × up_in → right. */
|
||||||
I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r_scratch
|
internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch
|
||||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||||
, U4 r_d /* load b.x */
|
, U4 r_d /* load b.x */
|
||||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
|
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
|
||||||
) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, ab, {
|
) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, aa, {
|
||||||
/* Compute the three scratch pointers from r_scratch. */
|
/* FIX: build packed RT22+RT33 with proper sign extension. */
|
||||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
|
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
|
||||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
|
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
|
||||||
@@ -312,24 +276,51 @@ I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r
|
|||||||
load_word(r_c, r_g, O_(V3_S4,z)),
|
load_word(r_c, r_g, O_(V3_S4,z)),
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0
|
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
|
||||||
(hardcoded; reusing the body's last two loads is fine because the load-delay slot is the nop after the third load,
|
|
||||||
and mtc2 below doesn't read these regs). */
|
|
||||||
load_word(r_d, r_h, O_(V3_S4,x)),
|
load_word(r_d, r_h, O_(V3_S4,x)),
|
||||||
load_word(R_AT, r_h, O_(V3_S4,y)),
|
load_word(R_AT, r_h, O_(V3_S4,y)),
|
||||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */
|
/* Save the two RT control-register slots OP will clobber. We reuse
|
||||||
gte_mv_to_data_r(r_a, C2_IR1),
|
* r_g/r_h (scratch pointers, no longer needed) as the save targets. */
|
||||||
gte_mv_to_data_r(r_b, C2_IR2),
|
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
|
||||||
gte_mv_to_data_r(r_c, C2_IR3),
|
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
|
||||||
gte_mv_to_data_r(r_d, C2_VXY0), /* D1 = b.x */
|
|
||||||
gte_mv_to_data_r(R_AT, C2_VZ0), /* D2 = b.y */
|
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY1), /* D3 = b.z */
|
|
||||||
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
|
||||||
|
|
||||||
gte_cmdw_outer_product, /* OP fires; MAC1/2/3 = a × b */
|
/* Load uz.x/uz.y/uz.z into COP2 control registers.
|
||||||
|
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
||||||
|
* RT22 is in BOTH $2.high AND $4.low (shared bit position). OP reads from $2.high.
|
||||||
|
* So set RT22 via ctc2 r_b, $2 (sets $2.high = a.y.high = RT22, $2.low = a.y.low = RT13).
|
||||||
|
* Then set RT33 via ctc2 r_c, $4 (sets $4.high = a.z.high = RT33, $4.low = a.z.low).
|
||||||
|
* The $2 and $4 writes don't clobber each other (separate registers).
|
||||||
|
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
|
||||||
|
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
|
||||||
|
* This is libpsyx's OuterProduct12 convention EXACTLY. */
|
||||||
|
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
|
||||||
|
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
|
||||||
|
|
||||||
|
/* Load uz into the RT diagonal. */
|
||||||
|
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
|
||||||
|
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
/* Load up_in into IR (the second operand for OP). */
|
||||||
|
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = up_in.x */
|
||||||
|
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
|
||||||
|
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = up_in.z */
|
||||||
|
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
gte_cmdw_outer_product, /* OP: MAC1/2/3 = uz × up_in
|
||||||
|
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
|
||||||
|
* MAC2 = IR1*D3 - IR3*D1 = up_in.x*uz.z.high - up_in.z*uz.x
|
||||||
|
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
|
||||||
|
* For up_in = (0, -fp_one, 0):
|
||||||
|
* MAC1 = 0 - (-fp_one)*uz.z.high = fp_one*uz.z.high
|
||||||
|
* MAC2 = 0 - 0 = 0
|
||||||
|
* MAC3 = (-fp_one)*uz.x - 0 = -fp_one*uz.x */
|
||||||
|
|
||||||
|
/* Restore the RT slots we clobbered. */
|
||||||
|
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
|
||||||
|
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
|
||||||
|
|
||||||
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
|
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
|
||||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||||
@@ -337,6 +328,12 @@ I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r
|
|||||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||||
nop, /* MFC2 retirement */
|
nop, /* MFC2 retirement */
|
||||||
|
|
||||||
|
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
|
||||||
|
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
|
||||||
|
shift_aright(r_a, r_a, 12),
|
||||||
|
shift_aright(r_b, r_b, 12),
|
||||||
|
shift_aright(r_c, r_c, 12),
|
||||||
|
|
||||||
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
|
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
|
||||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||||
@@ -346,15 +343,15 @@ I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r
|
|||||||
})
|
})
|
||||||
|
|
||||||
/* Atom 4: cross uz × ux → up. */
|
/* Atom 4: cross uz × ux → up. */
|
||||||
I_ void resolve_look_at__cross_uz_ux_to_up_proc(MipsAtomBuilder_R ab, U4 r_scratch
|
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
|
||||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||||
, U4 r_d /* load b.x */
|
, U4 r_d /* load b.x */
|
||||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
|
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
|
||||||
) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, ab, {
|
) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, aa, {
|
||||||
/* Compute the three scratch pointers from r_scratch. */
|
/* Compute the three scratch pointers from r_scratch. */
|
||||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
|
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
|
||||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
|
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||||
@@ -369,20 +366,47 @@ I_ void resolve_look_at__cross_uz_ux_to_up_proc(MipsAtomBuilder_R ab, U4 r_scrat
|
|||||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* mtc2 a → IR1/2/3, b → D1/2/3 (VXY0/VZ0/VXY1). */
|
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
|
||||||
gte_mv_to_data_r(r_a, C2_IR1),
|
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
|
||||||
gte_mv_to_data_r(r_b, C2_IR2),
|
* ctc2 RT restore. */
|
||||||
gte_mv_to_data_r(r_c, C2_IR3),
|
|
||||||
gte_mv_to_data_r(r_d, C2_VXY0),
|
/* Save the two RT control-register slots OP will clobber (reusing
|
||||||
gte_mv_to_data_r(R_AT, C2_VZ0),
|
* r_g/r_h — they're no longer needed as scratch pointers). */
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY1),
|
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
|
||||||
nop2,
|
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
|
||||||
|
|
||||||
|
/* Load uz into the RT diagonal — same packing as atom 1.
|
||||||
|
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
||||||
|
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
|
||||||
|
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
|
||||||
|
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
|
||||||
|
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
|
||||||
|
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
|
||||||
|
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
|
||||||
|
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
|
||||||
|
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
/* Load ux into the IR registers (the second operand for OP). */
|
||||||
|
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
|
||||||
|
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
|
||||||
|
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
|
||||||
|
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
gte_cmdw_outer_product,
|
gte_cmdw_outer_product,
|
||||||
|
|
||||||
|
/* Restore the RT slots we clobbered. */
|
||||||
|
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
|
||||||
|
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
|
||||||
|
|
||||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||||
gte_mv_from_data_r(r_b, C2_MAC2),
|
gte_mv_from_data_r(r_b, C2_MAC2),
|
||||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||||
nop,
|
nop,
|
||||||
|
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
|
||||||
|
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
|
||||||
|
shift_aright(r_a, r_a, 12),
|
||||||
|
shift_aright(r_b, r_b, 12),
|
||||||
|
shift_aright(r_c, r_c, 12),
|
||||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||||
store_word(r_c, r_f, O_(V3_S4,z)),
|
store_word(r_c, r_f, O_(V3_S4,z)),
|
||||||
@@ -415,21 +439,20 @@ typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
|
|||||||
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
|
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
|
||||||
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
|
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
|
||||||
*/
|
*/
|
||||||
I_ void resolve_look_at__populate_and_translate_proc(MipsAtomBuilder_R ab
|
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
|
||||||
, U4 r_look_at
|
, U4 r_look_at
|
||||||
, U4 r_scratch
|
, U4 r_scratch
|
||||||
, U4 r_pux, U4 r_puy, U4 r_puz, U4 r_peye /* 4 dedicated pointer regs */
|
, U4 r_pux, U4 r_puy, U4 r_puz
|
||||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2 /* 3 atom-local scratch regs */
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||||
) MipsAtom_Proc_(resolve_look_at__populate_and_translate, ab, {
|
) MipsAtom_Proc_(resolve_look_at__populate, aa, {
|
||||||
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
|
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
|
||||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||||
|
|
||||||
/* Compute the 4 scratch pointers in their dedicated GPRs. */
|
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
|
||||||
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
|
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
|
||||||
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
|
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
|
||||||
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
|
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
|
||||||
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)), /* r_peye = &eye */
|
|
||||||
nop,
|
nop,
|
||||||
|
|
||||||
/* ── m[0] = (S2)ux ── */
|
/* ── m[0] = (S2)ux ── */
|
||||||
@@ -459,8 +482,66 @@ I_ void resolve_look_at__populate_and_translate_proc(MipsAtomBuilder_R ab
|
|||||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
|
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
|
||||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
|
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
|
||||||
|
|
||||||
/* ── Translation column t[i] = R * (-eye) ─────────────────────────────
|
/* Zero t[0..2] — atom 6c writes the final values here. */
|
||||||
* pos = -eye: load eye.x/y/z from r_peye, negate via sub_u from R_0. */
|
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
|
||||||
|
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
|
||||||
|
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
|
||||||
|
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
|
||||||
|
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
|
||||||
|
* Stores off to scratch+96 (overwriting the packed pos).
|
||||||
|
*
|
||||||
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
|
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
||||||
|
* r_peye : pointer to eye (slot +96, reused as off destination)
|
||||||
|
* r_tmp0/1/2: -eye + GTE transfer scratch
|
||||||
|
*
|
||||||
|
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
|
||||||
|
*/
|
||||||
|
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
|
||||||
|
, U4 r_scratch
|
||||||
|
, U4 r_peye
|
||||||
|
, U4 r_look_at
|
||||||
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||||
|
) MipsAtom_Proc_(resolve_look_at__matrix_vector, aa, {
|
||||||
|
/* === EXACT C11 ApplyMatrixLV replication ===
|
||||||
|
* The C11 does:
|
||||||
|
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
||||||
|
* 2. lw v.x/y/z from memory
|
||||||
|
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
|
||||||
|
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3)
|
||||||
|
* 5. mfc2 MACs
|
||||||
|
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
||||||
|
* 7. mfc2 MACs
|
||||||
|
* 8. Combine: (pass1 << 3) + pass2
|
||||||
|
*
|
||||||
|
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
|
||||||
|
* The combine simplifies: result = 0 + pass2 = pass2.
|
||||||
|
* So we skip the S15 decomposition and just do pass 2 directly.
|
||||||
|
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
|
||||||
|
|
||||||
|
/* Pop look_at* from tape. */
|
||||||
|
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||||
|
|
||||||
|
/* r_peye = &eye (slot +96, reused as off destination). */
|
||||||
|
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
|
||||||
|
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
|
||||||
|
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
|
||||||
|
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
|
||||||
|
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
|
||||||
|
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
|
||||||
|
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
|
||||||
|
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
|
||||||
|
|
||||||
|
/* Load pos = -eye after the matrix load releases r_tmp0. */
|
||||||
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
|
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
|
||||||
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
|
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
|
||||||
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
|
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
|
||||||
@@ -469,27 +550,63 @@ I_ void resolve_look_at__populate_and_translate_proc(MipsAtomBuilder_R ab
|
|||||||
sub_u(r_tmp1, R_0, r_tmp1),
|
sub_u(r_tmp1, R_0, r_tmp1),
|
||||||
sub_u(r_tmp2, R_0, r_tmp2),
|
sub_u(r_tmp2, R_0, r_tmp2),
|
||||||
|
|
||||||
/* mtc2 IR1/2/3 = pos (for MVMVA — input vector registers). */
|
/* === mtc2 pos (as S16) to IR1/2/3 ===
|
||||||
|
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
|
||||||
|
* 32-bit sign-extended value's low 16 bits = correct S16. */
|
||||||
|
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
|
||||||
|
* gives the correct S16 value (sign bit preserved). */
|
||||||
|
/* r_tmp0/1/2 already have pos values. */
|
||||||
gte_mv_to_data_r(r_tmp0, C2_IR1),
|
gte_mv_to_data_r(r_tmp0, C2_IR1),
|
||||||
gte_mv_to_data_r(r_tmp1, C2_IR2),
|
gte_mv_to_data_r(r_tmp1, C2_IR2),
|
||||||
gte_mv_to_data_r(r_tmp2, C2_IR3),
|
gte_mv_to_data_r(r_tmp2, C2_IR3),
|
||||||
nop2,
|
nop2, /* MTC2 retirement (2 slots) */
|
||||||
|
|
||||||
/* MVMVA: MAC1/2/3 = R * IR with cv=0 (no TR vector), mx=0 (rotation matrix), sf=0 (no shift), v=0 (V0 = IR1/2/3, no far-plane clipping).
|
/* === MVMVA pass 2 EXACT C11 command: 0x4A49E012 ===
|
||||||
* The pre-set rotation matrix is the one set by the preceding set_gte_world atom.
|
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
||||||
* gte_cmdw_mvmva is parameterless and defaults to cv=0/mx=0/sf=0/v=0. */
|
gte_cmdw_mvmva_c11_pass2_exact,
|
||||||
gte_cmdw_mvmva,
|
|
||||||
nop, /* GTE interlock */
|
nop, /* GTE interlock */
|
||||||
|
|
||||||
/* mfc2 MAC1/2/3 → r_tmp0/r_tmp1/r_tmp2 (sign-extended into 32-bit GPRs).
|
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
|
||||||
* MAC1/2/3 hold R*v with no TR add and no perspective divide — exactly the 3 distinct world-space translation values we need for t[0..2]. */
|
|
||||||
gte_mv_from_data_r(r_tmp0, C2_MAC1),
|
gte_mv_from_data_r(r_tmp0, C2_MAC1),
|
||||||
gte_mv_from_data_r(r_tmp1, C2_MAC2),
|
gte_mv_from_data_r(r_tmp1, C2_MAC2),
|
||||||
gte_mv_from_data_r(r_tmp2, C2_MAC3),
|
gte_mv_from_data_r(r_tmp2, C2_MAC3),
|
||||||
nop,
|
nop,
|
||||||
store_word(r_tmp0, r_look_at, O_(MT3_S2S4,t[0])),
|
|
||||||
store_word(r_tmp1, r_look_at, O_(MT3_S2S4,t[1])),
|
/* === Store off → scratch+96 (overwriting pos) === */
|
||||||
store_word(r_tmp2, r_look_at, O_(MT3_S2S4,t[2])),
|
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
|
||||||
|
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
|
||||||
|
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Atom 6c in the bundle: copy scratch+96 (off, written by atom 6b) → look_at->t[].
|
||||||
|
* Uses mac_trans_matrix component (m->t = v, libgte TransMatrix semantics = struct copy).
|
||||||
|
*
|
||||||
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
|
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
||||||
|
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
||||||
|
* r_off_ptr : pointer to off (= &scratch.eye, reused slot)
|
||||||
|
* r_tmp0 : transfer reg for mac_trans_matrix
|
||||||
|
*
|
||||||
|
* Pool cost: r_look_at (1) + r_scratch (carrier) + r_off_ptr + 1 clobber = 4 GPRs.
|
||||||
|
*/
|
||||||
|
I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa
|
||||||
|
, U4 r_look_at
|
||||||
|
, U4 r_scratch
|
||||||
|
, U4 r_off_ptr
|
||||||
|
, U4 r_tmp0
|
||||||
|
) MipsAtom_Proc_(resolve_look_at__trans_matrix, aa, {
|
||||||
|
/* Pop look_at* from tape. */
|
||||||
|
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||||
|
|
||||||
|
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */
|
||||||
|
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */
|
||||||
|
mac_trans_matrix(r_look_at, r_off_ptr, r_tmp0),
|
||||||
|
|
||||||
mac_yield()
|
mac_yield()
|
||||||
})
|
})
|
||||||
@@ -512,47 +629,47 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
|||||||
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||||
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + OA_(DoubleBuffer,display,0)),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,0)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,0)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,0)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
|
||||||
|
|
||||||
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + OA_(DoubleBuffer,display,1)),
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[1])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + OA_(DoubleBuffer,display,1)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[1])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[1])),
|
||||||
|
|
||||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + O_(DoubleBuffer,draw[0])), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + O_(DoubleBuffer,draw[0])), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,0)),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[0])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,0)),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[0])),
|
||||||
|
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + OA_(DoubleBuffer,draw,1)),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + O_(DoubleBuffer,draw[1])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + OA_(DoubleBuffer,draw,1)),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[1])),
|
||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||||
add_ui(R_T0, R_0, gp0_tpage_default),
|
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[0])),
|
||||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||||
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||||
add_ui(R_T0, R_0, 1),
|
add_ui(R_T0, R_0, 1),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,0)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[0])),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,0)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[0])),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,0)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[0])),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + OA_(DoubleBuffer,draw,1)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[1])),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + OA_(DoubleBuffer,draw,1)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[1])),
|
||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||||
add_ui(R_T0, R_0, 7),
|
add_ui(R_T0, R_0, 7),
|
||||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[0])),
|
||||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
mac_yield(),
|
mac_yield(),
|
||||||
};
|
};
|
||||||
|
|||||||
+232
-151
@@ -52,10 +52,16 @@
|
|||||||
#include "hello_camera.atom.c"
|
#include "hello_camera.atom.c"
|
||||||
#pragma endregion Hello Joypad TUs
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
|
enum {
|
||||||
|
Scratchpad_Loc = 0x1F800000,
|
||||||
|
};
|
||||||
|
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
Scratchpad_Len = 1024,
|
Scratchpad_Len = 1024,
|
||||||
MemTape_Len = 512,
|
MemTape_Len = 512,
|
||||||
ResolveLookAtArena_Words = 512,
|
ResolveLookAtArena_Words = 1024,
|
||||||
|
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
||||||
};
|
};
|
||||||
typedef Struct_(SMemory) {
|
typedef Struct_(SMemory) {
|
||||||
PrimitiveArena primitives;
|
PrimitiveArena primitives;
|
||||||
@@ -76,18 +82,11 @@ typedef Struct_(SMemory) {
|
|||||||
PadBiosRaw pad_raw[2];
|
PadBiosRaw pad_raw[2];
|
||||||
PadState pad[2];
|
PadState pad[2];
|
||||||
|
|
||||||
|
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
||||||
U4_V scratchpad; // d-cache
|
U4_V scratchpad; // d-cache
|
||||||
|
|
||||||
/* resolve_look_at bundle: pre-built atom arena + atom-refs.
|
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
||||||
* (Task 12.5 fix: moved from file-scope globals to smem fields.
|
MipsAtom* resolve_look_at_atom_addrs[10];
|
||||||
* Task 12.7 fix: dropped the ResolveLookAtScratch struct-as-view; the
|
|
||||||
* C-side helper uses `& smem.scratchpad[N]` at hardcoded offsets directly.
|
|
||||||
* Task 12.8 fix: chain atoms use r_scratch + offset internally; no C-side magic offsets anywhere.
|
|
||||||
* Task 12.11 fix: ResolveLookAtScratch offset schema moved to hello_camera.atom.c — gte.atom.c
|
|
||||||
* is the GENERIC GTE primitives file and must not know about the resolve_look_at bundle's scratch layout.) */
|
|
||||||
U4 resolve_look_at_arena[ResolveLookAtArena_Words]; /* ~2 KB; bumped from 420 per Task 4 subagent */
|
|
||||||
MipsAtom* resolve_look_at_atom_addrs[7];
|
|
||||||
MipsAtomBuilder resolve_look_at_ab_static;
|
|
||||||
};
|
};
|
||||||
global SMemory smem;
|
global SMemory smem;
|
||||||
extern SMemory smem;
|
extern SMemory smem;
|
||||||
@@ -105,8 +104,7 @@ I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
|||||||
}
|
}
|
||||||
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||||
|
|
||||||
void
|
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||||
resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
|
||||||
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
||||||
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
||||||
V3_S4 right, up, forward;
|
V3_S4 right, up, forward;
|
||||||
@@ -131,6 +129,7 @@ resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in)
|
|||||||
mul_m3s2_v3s4(look_at, & pos, & off);
|
mul_m3s2_v3s4(look_at, & pos, & off);
|
||||||
trans_m3s2( look_at, & off);
|
trans_m3s2( look_at, & off);
|
||||||
}
|
}
|
||||||
|
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||||
|
|
||||||
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
|
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
|
||||||
* Called ONCE from main() before the frame loop.
|
* Called ONCE from main() before the frame loop.
|
||||||
@@ -158,105 +157,215 @@ resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in)
|
|||||||
* GPR pool per atom: 10 free GPRs (R_T0..R_T3 + R_T5..R_T7 + R_V0 + R_V1 + R_AT).
|
* GPR pool per atom: 10 free GPRs (R_T0..R_T3 + R_T5..R_T7 + R_V0 + R_V1 + R_AT).
|
||||||
* R_T4 is reserved as the wave-context carrier (R_ResolveScratch).
|
* R_T4 is reserved as the wave-context carrier (R_ResolveScratch).
|
||||||
*/
|
*/
|
||||||
|
/* === EXPLICIT REGISTER ALLOCATION TRACKER ===
|
||||||
|
* Every GPR used by every atom is tracked below. NO GPR is assigned to
|
||||||
|
* two atoms at overlapping lifetimes. The tape runtime preserves R_T8/R_T9
|
||||||
|
* (R_AtomJmp/R_TapePtr) and clobbers R_T0-R_T7, R_AT, R_V0, R_V1.
|
||||||
|
* R_T4 is reserved as R_ResolveScratch (wave-context carrier).
|
||||||
|
*
|
||||||
|
* GPR pool: R_T0($8), R_T1($9), R_T2($10), R_T3($11), R_T5($13),
|
||||||
|
* R_T6($14), R_T7($15), R_V0($2), R_V1($3), R_AT($1)
|
||||||
|
* Reserved: R_T4($12) = R_ResolveScratch
|
||||||
|
* Tape: R_T8($24) = R_AtomJmp, R_T9($25) = R_TapePtr (preserved)
|
||||||
|
*
|
||||||
|
* === ATOM 0: input_and_sub (stages eye/up_in, computes fwd) ===
|
||||||
|
* Pop tape → R_T0(target), R_T1(eye), R_T2(up_in).
|
||||||
|
* Use R_T3,R_T5,R_T6,R_T7 as temps.
|
||||||
|
* NO conflict with other atoms (each atom has independent lifetime).
|
||||||
|
*
|
||||||
|
* === ATOM 1: normalize fwd→uz ===
|
||||||
|
* r_src_offset=0, r_dst_offset=16.
|
||||||
|
* r_src_ptr=R_T0, r_dst_ptr=R_T1, r_tmp=R_T2 (preserved for stage 4).
|
||||||
|
* r_mac1=R_T3, r_mac2=R_T5, r_recip=R_T6, r_lzcr=R_T7, r_shift=R_V0, r_branch=R_V1.
|
||||||
|
*
|
||||||
|
* === ATOM 2: cross uz×up_in→right ===
|
||||||
|
* r_a=R_T0, r_b=R_T1, r_c=R_T2, r_d=R_T3, r_f(out)=R_T5, r_g=R_T6, r_h=R_T7.
|
||||||
|
*
|
||||||
|
* === ATOM 3: normalize right→ux ===
|
||||||
|
* Same GPR pool as atom 1.
|
||||||
|
*
|
||||||
|
* === ATOM 4: cross uz×ux→up ===
|
||||||
|
* r_a=R_T0, r_b=R_T1, r_c=R_T2, r_d=R_T3, r_f(out)=R_T5, r_g=R_T6, r_h=R_T7.
|
||||||
|
*
|
||||||
|
* === ATOM 5: normalize up→uy ===
|
||||||
|
* Same GPR pool as atom 1.
|
||||||
|
*
|
||||||
|
* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) ===
|
||||||
|
* r_look_at=R_T0 (pop tape), r_scratch=R_T4.
|
||||||
|
* r_pux=R_T1, r_puy=R_T3, r_puz=R_T5.
|
||||||
|
* r_tmp0=R_T2, r_tmp1=R_T6, r_tmp2=R_V0.
|
||||||
|
*
|
||||||
|
* === ATOM 6a.5: set_gte_mt3s2s4 (ctc2 RT matrix) ===
|
||||||
|
* BAKED atom. Uses R_T3 internally (hardcoded in gte.atom.c).
|
||||||
|
* NO conflict — different GPR pool, and the atom body hardcodes R_T3
|
||||||
|
* as the matrix pointer. We DON'T need to assign R_T3 to atom 6a.5
|
||||||
|
* because it's a baked atom with its own GPR usage.
|
||||||
|
*
|
||||||
|
* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
|
||||||
|
* r_look_at=R_T0 (pop tape), r_scratch=R_T4.
|
||||||
|
* r_peye=R_T1.
|
||||||
|
* r_tmp0=R_T2, r_tmp1=R_T3, r_tmp2=R_T5.
|
||||||
|
* Uses mac_apply_matrix_lv which internally uses these temps.
|
||||||
|
*
|
||||||
|
* === ATOM 6c: trans_matrix (off → look_at->t[]) ===
|
||||||
|
* r_look_at=R_T0 (pop tape), r_scratch=R_T4.
|
||||||
|
* r_off_ptr=R_T1.
|
||||||
|
* r_tmp0=R_T2.
|
||||||
|
*
|
||||||
|
* === CONFLICT CHECK ===
|
||||||
|
* All atoms use the same GPR pool R_T0-R_T3, R_T5-R_T7, R_V0-R_V1.
|
||||||
|
* But atoms are SEQUENTIAL — each atom's lifetime is disjoint from
|
||||||
|
* the next atom's lifetime. The tape yield handshake between atoms
|
||||||
|
* preserves R_TapePtr (R_T9) and R_AtomJmp (R_T8).
|
||||||
|
*
|
||||||
|
* The GPR pool is SHARED across atoms (they run sequentially, not
|
||||||
|
* concurrently). Each atom's build call assigns specific R_T* codes
|
||||||
|
* for that atom's body. The same R_T* code can be reused across atoms
|
||||||
|
* because the previous atom's body has already yielded.
|
||||||
|
*/
|
||||||
internal void resolve_look_at_init(void) {
|
internal void resolve_look_at_init(void) {
|
||||||
/* Wrap the static arena in a MipsAtomBuilder. */
|
/* Wrap the static arena in a MipsAtomBuilder. */
|
||||||
MipsAtomBuilder_R ab = & smem.resolve_look_at_ab_static;
|
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||||
ab->start = u4_(smem.resolve_look_at_arena);
|
|
||||||
ab->capacity = ResolveLookAtArena_Words;
|
|
||||||
ab->used = 0;
|
|
||||||
|
|
||||||
/* Atom 0: resolve_look_at__input_and_sub — stages eye/up_in into scratchpad,
|
/* === ATOM 0: input_and_sub === */
|
||||||
* computes fwd = target - eye; binds R_ResolveScratch (R_T4) as the wave-context carrier for atoms 1-6.
|
U4 const r_target_ptr = R_T0; /* tape pop → target */
|
||||||
* The body hardcodes R_AT and R_V0 as eye.y/eye.z temps (the existing sub_u(eye.x, eye.y, eye.z) chain from the prior Task 12.7 design). */
|
U4 const r_eye_ptr = R_T1; /* tape pop → eye */
|
||||||
smem.resolve_look_at_atom_addrs[0] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
U4 const r_up_in_ptr = R_T2; /* tape pop → up_in */
|
||||||
resolve_look_at__input_and_sub_proc(ab, R_ResolveScratch,
|
U4 const r_tmp0_0 = R_T3;
|
||||||
R_T0, /* r_target_ptr (popped from tape) */
|
U4 const r_tmp1_0 = R_T5;
|
||||||
R_T1, /* r_eye_ptr (popped from tape) */
|
U4 const r_tmp2_0 = R_T6;
|
||||||
R_T2, /* r_up_in_ptr (popped from tape) */
|
U4 const r_tmp3_0 = R_T7;
|
||||||
R_T3, R_T5, R_T6, R_T7); /* r_tmp<0-3> */
|
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_target_ptr, r_eye_ptr, r_up_in_ptr,
|
||||||
|
r_tmp0_0, r_tmp1_0, r_tmp2_0, r_tmp3_0);
|
||||||
|
|
||||||
/* Atom 1: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+0=fwd, dst=scratch+16=uz.
|
/* === ATOM 1: normalize fwd→uz === */
|
||||||
* The proc takes r_src_offset + r_dst_offset as U4 PARAMETERS — we pass the O_(...) macros here (evaluating to numeric literals 0 and 16).
|
U4 const r_src_offset_1 = O_(ResolveLookAtScratch, fwd);
|
||||||
* The 4-stage body is identical across the 3 call sites (atoms 1, 3, 5); only the offset args differ.
|
U4 const r_dst_offset_1 = O_(ResolveLookAtScratch, uz);
|
||||||
* GPR pool: r_scratch (R_T4 carrier) + 9 body GPRs = 10.
|
U4 const r_src_ptr_1 = R_T0;
|
||||||
* r_src_ptr (R_T0) : src ptr
|
U4 const r_dst_ptr_1 = R_T1;
|
||||||
* r_dst_ptr (R_T1) : dst ptr
|
U4 const r_tmp_1 = R_T2;
|
||||||
* r_tmp (R_T2) : unused (reserved for symmetry)
|
U4 const r_mac1_1 = R_T3;
|
||||||
* r_mac1_scratch (R_T3) : MAC1 scratch
|
U4 const r_mac2_1 = R_T5;
|
||||||
* r_mac2_scratch (R_T5) : src.x → result.x (carries through stages 1-2)
|
U4 const r_recip_1 = R_T6;
|
||||||
* r_recip_est (R_T6) : src.y → result.y
|
U4 const r_lzcr_1 = R_T7;
|
||||||
* r_lzcr (R_T7) : |v|² accumulator + srav amount (single reg)
|
U4 const r_shift_1 = R_V0;
|
||||||
* r_shift (R_V0) : LZCR (saved across stages 3-4)
|
U4 const r_branch_1 = R_V1;
|
||||||
* r_branch_tmp (R_V1) : src.z → result.z (reused after stage 1)
|
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
|
||||||
*/
|
R_ResolveScratch,
|
||||||
smem.resolve_look_at_atom_addrs[1] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
r_src_offset_1, r_dst_offset_1,
|
||||||
normalize_v3s4_proc(ab, R_ResolveScratch, /* r_scratch (wave-context carrier) */
|
r_src_ptr_1, r_dst_ptr_1, r_tmp_1,
|
||||||
O_(ResolveLookAtScratch, fwd), /* r_src_offset = 0 */
|
r_mac1_1, r_mac2_1, r_recip_1, r_lzcr_1,
|
||||||
O_(ResolveLookAtScratch, uz), /* r_dst_offset = 16 */
|
r_shift_1, r_branch_1);
|
||||||
R_T0, R_T1, R_T2, /* r_src_ptr, r_dst_ptr, r_tmp */
|
|
||||||
R_T3, /* r_mac1_scratch */
|
|
||||||
R_T5, /* r_mac2_scratch */
|
|
||||||
R_T6, /* r_recip_est */
|
|
||||||
R_T7, /* r_lzcr */
|
|
||||||
R_V0, /* r_shift */
|
|
||||||
R_V1); /* r_branch_tmp */
|
|
||||||
|
|
||||||
/* Atom 2: resolve_look_at__cross_uz_up_in_to_right — a=scratch+16, b=scratch+128,
|
/* === ATOM 2: cross uz×up_in→right === */
|
||||||
* out=scratch+32 (HARDCODED in body). GPR pool: r_scratch + 7 body + R_AT + R_V0 = 10. */
|
U4 const r_a_2 = R_T0;
|
||||||
smem.resolve_look_at_atom_addrs[2] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
U4 const r_b_2 = R_T1;
|
||||||
resolve_look_at__cross_uz_up_in_to_right_proc(ab, R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */
|
U4 const r_c_2 = R_T2;
|
||||||
R_T0, R_T1, R_T2, /* r_a, r_b, r_c (a.x/y/z → out.x/y/z) */
|
U4 const r_d_2 = R_T3;
|
||||||
R_T3, /* r_d (b.x) */
|
U4 const r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
|
||||||
R_T5, /* r_f (out ptr = scratch+32) */
|
U4 const r_g_2 = R_T6; /* a ptr = scratch+16 */
|
||||||
R_T6, /* r_g (a ptr = scratch+16) */
|
U4 const r_h_2 = R_T7; /* b ptr = scratch+128 */
|
||||||
R_T7); /* r_h (b ptr = scratch+128) */
|
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
|
||||||
|
|
||||||
/* Atom 3: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+32=right, dst=scratch+48=ux. */
|
/* === ATOM 3: normalize right→ux === */
|
||||||
smem.resolve_look_at_atom_addrs[3] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
U4 const r_src_offset_3 = O_(ResolveLookAtScratch, right);
|
||||||
normalize_v3s4_proc(ab, R_ResolveScratch,
|
U4 const r_dst_offset_3 = O_(ResolveLookAtScratch, ux);
|
||||||
O_(ResolveLookAtScratch, right), /* r_src_offset = 32 */
|
U4 const r_src_ptr_3 = R_T0;
|
||||||
O_(ResolveLookAtScratch, ux), /* r_dst_offset = 48 */
|
U4 const r_dst_ptr_3 = R_T1;
|
||||||
R_T0, R_T1, R_T2,
|
U4 const r_tmp_3 = R_T2;
|
||||||
R_T3,
|
U4 const r_mac1_3 = R_T3;
|
||||||
R_T5,
|
U4 const r_mac2_3 = R_T5;
|
||||||
R_T6,
|
U4 const r_recip_3 = R_T6;
|
||||||
R_T7,
|
U4 const r_lzcr_3 = R_T7;
|
||||||
R_V0,
|
U4 const r_shift_3 = R_V0;
|
||||||
R_V1);
|
U4 const r_branch_3 = R_V1;
|
||||||
|
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_src_offset_3, r_dst_offset_3,
|
||||||
|
r_src_ptr_3, r_dst_ptr_3, r_tmp_3,
|
||||||
|
r_mac1_3, r_mac2_3, r_recip_3, r_lzcr_3,
|
||||||
|
r_shift_3, r_branch_3);
|
||||||
|
|
||||||
/* Atom 4: resolve_look_at__cross_uz_ux_to_up — a=scratch+16, b=scratch+48, out=scratch+64 (HARDCODED). */
|
/* === ATOM 4: cross uz×ux→up === */
|
||||||
smem.resolve_look_at_atom_addrs[4] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
U4 const r_a_4 = R_T0;
|
||||||
resolve_look_at__cross_uz_ux_to_up_proc(ab, R_ResolveScratch,
|
U4 const r_b_4 = R_T1;
|
||||||
R_T0, R_T1, R_T2,
|
U4 const r_c_4 = R_T2;
|
||||||
R_T3,
|
U4 const r_d_4 = R_T3;
|
||||||
R_T5, /* r_f (out ptr = scratch+64) */
|
U4 const r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
|
||||||
R_T6, /* r_g (a ptr = scratch+16) */
|
U4 const r_g_4 = R_T6; /* a ptr = scratch+16 */
|
||||||
R_T7); /* r_h (b ptr = scratch+48) */
|
U4 const r_h_4 = R_T7; /* b ptr = scratch+48 */
|
||||||
|
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
|
||||||
|
|
||||||
/* Atom 5: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+64=up, dst=scratch+80=uy. */
|
/* === ATOM 5: normalize up→uy === */
|
||||||
smem.resolve_look_at_atom_addrs[5] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
U4 const r_src_offset_5 = O_(ResolveLookAtScratch, up);
|
||||||
normalize_v3s4_proc(ab, R_ResolveScratch,
|
U4 const r_dst_offset_5 = O_(ResolveLookAtScratch, uy);
|
||||||
O_(ResolveLookAtScratch, up), /* r_src_offset = 64 */
|
U4 const r_src_ptr_5 = R_T0;
|
||||||
O_(ResolveLookAtScratch, uy), /* r_dst_offset = 80 */
|
U4 const r_dst_ptr_5 = R_T1;
|
||||||
R_T0, R_T1, R_T2,
|
U4 const r_tmp_5 = R_T2;
|
||||||
R_T3,
|
U4 const r_mac1_5 = R_T3;
|
||||||
R_T5,
|
U4 const r_mac2_5 = R_T5;
|
||||||
R_T6,
|
U4 const r_recip_5 = R_T6;
|
||||||
R_T7,
|
U4 const r_lzcr_5 = R_T7;
|
||||||
R_V0,
|
U4 const r_shift_5 = R_V0;
|
||||||
R_V1);
|
U4 const r_branch_5 = R_V1;
|
||||||
|
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_src_offset_5, r_dst_offset_5,
|
||||||
|
r_src_ptr_5, r_dst_ptr_5, r_tmp_5,
|
||||||
|
r_mac1_5, r_mac2_5, r_recip_5, r_lzcr_5,
|
||||||
|
r_shift_5, r_branch_5);
|
||||||
|
|
||||||
/* Atom 6: resolve_look_at__populate_and_translate — write look_at->m[][] from ux/uy/uz (computed from r_scratch+offset internally),
|
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
|
||||||
then compute translation column t[] = R * (-eye). GPR pool: r_look_at + r_scratch + 4 ptr regs + 3 tmp regs = 9. */
|
U4 const r_look_at_6a = R_T0; /* tape pop → look_at* */
|
||||||
smem.resolve_look_at_atom_addrs[6] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4));
|
U4 const r_scratch_6a = R_ResolveScratch;
|
||||||
resolve_look_at__populate_and_translate_proc(ab,
|
U4 const r_pux_6a = R_T1;
|
||||||
R_T0, /* r_look_at (popped from tape; MT3_S2S4*) */
|
U4 const r_puy_6a = R_T3;
|
||||||
R_ResolveScratch, /* r_scratch (wave-context carrier) */
|
U4 const r_puz_6a = R_T5;
|
||||||
R_T1, R_T3, R_T5, R_T7, /* r_pux, r_puy, r_puz, r_peye */
|
U4 const r_tmp0_6a = R_T2;
|
||||||
R_T2, R_T6, R_V0); /* r_tmp0, r_tmp1, r_tmp2 */
|
U4 const r_tmp1_6a = R_T6;
|
||||||
|
U4 const r_tmp2_6a = R_V0;
|
||||||
|
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
|
||||||
|
r_look_at_6a, r_scratch_6a,
|
||||||
|
r_pux_6a, r_puy_6a, r_puz_6a,
|
||||||
|
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
|
||||||
|
|
||||||
|
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
|
||||||
|
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
|
||||||
|
* the matrix pointer (popped from tape). It does NOT need GPR
|
||||||
|
* assignment from us — it has its own internal GPR usage.
|
||||||
|
* We just take its address. */
|
||||||
|
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
|
||||||
|
|
||||||
|
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
|
||||||
|
* Uses mac_apply_matrix_lv component macro which internally uses
|
||||||
|
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
|
||||||
|
* for the mfc2/store. We pass our GPRs. */
|
||||||
|
U4 const r_scratch_6b = R_ResolveScratch;
|
||||||
|
U4 const r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
|
||||||
|
U4 const r_look_at_6b = R_T0; /* tape pop → look_at* */
|
||||||
|
U4 const r_tmp0_6b = R_T2;
|
||||||
|
U4 const r_tmp1_6b = R_T3;
|
||||||
|
U4 const r_tmp2_6b = R_T5;
|
||||||
|
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
|
||||||
|
r_scratch_6b, r_peye_6b, r_look_at_6b,
|
||||||
|
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
|
||||||
|
|
||||||
|
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
|
||||||
|
U4 const r_look_at_6c = R_T0; /* tape pop → look_at* */
|
||||||
|
U4 const r_scratch_6c = R_ResolveScratch;
|
||||||
|
U4 const r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
|
||||||
|
U4 const r_tmp0_6c = R_T2;
|
||||||
|
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
|
||||||
|
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c);
|
||||||
|
|
||||||
/* Sanity check: arena didn't overflow. */
|
/* Sanity check: arena didn't overflow. */
|
||||||
assert(ab->used <= ResolveLookAtArena_Words);
|
assert(ab.used <= ResolveLookAtArena_Size);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
||||||
@@ -277,30 +386,33 @@ I_ void resolve_look_at(
|
|||||||
, P3_S4* target
|
, P3_S4* target
|
||||||
, V3_S4* up_in
|
, V3_S4* up_in
|
||||||
){
|
){
|
||||||
/* Atom 0: input_and_sub — stages eye/up_in into scratchpad + computes fwd. */
|
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
|
||||||
tb_data(tb, u4_(target)); /* Binds_ResolveLookAtSub.target (C-side P3_S4*) */
|
tb_data(tb, u4_(target));
|
||||||
tb_data(tb, u4_(eye)); /* Binds_ResolveLookAtSub.eye (C-side P3_S4*) */
|
tb_data(tb, u4_(eye));
|
||||||
tb_data(tb, u4_(up_in)); /* Binds_ResolveLookAtSub.up_in (C-side V3_S4*) */
|
tb_data(tb, u4_(up_in));
|
||||||
tb_data(tb, u4_(smem.scratchpad)); /* Binds_ResolveLookAtScratch.scratch_base */
|
tb_data(tb, u4_(smem.scratchpad));
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Atoms 1-5: NO tb_data — each chain atom uses r_scratch + hardcoded_offset internally (no tape-data pointers between atoms).
|
|
||||||
Context carrier R_ResolveScratch (R_T4) is preserved across atoms. */
|
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
|
||||||
|
|
||||||
/* Atom 6: populate_and_translate — only output pointer is the matrix destination. */
|
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
|
||||||
tb_data(tb, u4_(look_at)); /* Binds_ResolveLookAtPopAndTrans.look_at (MT3_S2S4*) */
|
tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
|
||||||
|
tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
|
||||||
|
tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
|
||||||
|
tb_data(tb, u4_(look_at));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
|
||||||
|
|
||||||
GCC_OPTIMIZATION_DISABLE
|
GCC_OPTIMIZATION_DISABLE
|
||||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||||
{
|
{
|
||||||
@@ -351,48 +463,15 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
A2_S2 p; //???
|
A2_S2 p; //???
|
||||||
S4 flag; //????
|
S4 flag; //????
|
||||||
|
|
||||||
// Camera Look at (Tape) + inline C11 fallback — bundle runs, then C11 inlines the look_at.
|
B4 use_c11_path = false;
|
||||||
// Currently: bundle's atom 0 (input_and_sub) runs + C11 does the rest. As bundle atoms
|
if (use_c11_path) {
|
||||||
// are incrementally fixed, the corresponding C11 lines get commented out.
|
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||||
if (1)
|
}
|
||||||
|
if (use_c11_path == false)
|
||||||
{
|
{
|
||||||
tb.used = 0; tb_scope_run(& tb) {
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||||
}
|
}
|
||||||
|
|
||||||
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
|
||||||
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
|
||||||
V3_S4 right, up, forward;
|
|
||||||
V3_S4 ux, uy, uz;
|
|
||||||
V3_S4 pos, off;
|
|
||||||
|
|
||||||
// forward = smem.cube.pos; sub_v3s4(& forward, smem.cam.pos); // RGA(Lengyel): Affine point - point = zero-weight direction. (now done by bundle atom 0)
|
|
||||||
// Read fwd from scratchpad[+0] (atom 0's output)
|
|
||||||
forward.x = u4_v(0x1F800000)[0];
|
|
||||||
forward.y = u4_v(0x1F800000)[1];
|
|
||||||
forward.z = u4_v(0x1F800000)[2];
|
|
||||||
forward.pad = u4_v(0x1F800000)[3];
|
|
||||||
// normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization. (now done by bundle atom 1)
|
|
||||||
// Read uz from scratchpad[+16] (atom 1's output)
|
|
||||||
uz.x = u4_v(0x1F800010)[0];
|
|
||||||
uz.y = u4_v(0x1F800010)[1];
|
|
||||||
uz.z = u4_v(0x1F800010)[2];
|
|
||||||
uz.pad = u4_v(0x1F800010)[3];
|
|
||||||
|
|
||||||
cross_v3s4(& uz, & v3s4(0, -fp_one, 0), & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
|
||||||
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
|
||||||
|
|
||||||
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
|
||||||
smem.cam.look_at.m[0][0] = ux.x; smem.cam.look_at.m[0][1] = ux.y; smem.cam.look_at.m[0][2] = ux.z;
|
|
||||||
smem.cam.look_at.m[1][0] = uy.x; smem.cam.look_at.m[1][1] = uy.y; smem.cam.look_at.m[1][2] = uy.z;
|
|
||||||
smem.cam.look_at.m[2][0] = uz.x; smem.cam.look_at.m[2][1] = uz.y; smem.cam.look_at.m[2][2] = uz.z;
|
|
||||||
|
|
||||||
pos = smem.cam.pos; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
|
|
||||||
|
|
||||||
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
|
|
||||||
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
|
|
||||||
mul_m3s2_v3s4(& smem.cam.look_at, & pos, & off);
|
|
||||||
trans_m3s2( & smem.cam.look_at, & off);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Draw cube
|
// Draw cube
|
||||||
@@ -498,7 +577,8 @@ GCC_OPTIMIZATION_DISABLE
|
|||||||
int main(void)
|
int main(void)
|
||||||
{
|
{
|
||||||
smem = (SMemory){0};
|
smem = (SMemory){0};
|
||||||
smem.scratchpad = C_(U4_V, 0x1F800000);
|
// TODO(Ed): remove this field we don't need it in smem.
|
||||||
|
smem.scratchpad = C_(U4_V, Scratchpad_Loc);
|
||||||
// smem.primitives.used = 0;
|
// smem.primitives.used = 0;
|
||||||
// smem.active_buf_id = 0;
|
// smem.active_buf_id = 0;
|
||||||
smem.cam.pos = v3s4(500, -1000, -1500);
|
smem.cam.pos = v3s4(500, -1000, -1500);
|
||||||
@@ -544,3 +624,4 @@ int main(void)
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
GCC_OPTIMIZATION_ENABLE
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|
||||||
|
|||||||
+10625
File diff suppressed because one or more lines are too long
@@ -1313,6 +1313,21 @@ M.GTE_COMMAND_LATCH_WINDOWS = {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
--- GTE control-register alias groups.
|
||||||
|
--- Aliases within a group write to the same C2 control-register slot on real silicon
|
||||||
|
--- (the silicon double-maps some C2 slots across multiple PSX SDK / libgte conventions).
|
||||||
|
--- Aliases across groups write to distinct C2 slots.
|
||||||
|
---
|
||||||
|
--- Cross-alias writes inside one atom body, or across the wave-context boundary,
|
||||||
|
--- silently clobber each other. The `check_gte_cr_alias_writes` check warns about
|
||||||
|
--- each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
|
||||||
|
--- for the silicon rationale and the libgte outer-product convention.
|
||||||
|
M.GTE_CR_ALIAS_GROUPS = {
|
||||||
|
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
|
||||||
|
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
|
||||||
|
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
|
||||||
|
}
|
||||||
|
|
||||||
-- Operand-class table for the COP2->GPR load-delay check.
|
-- Operand-class table for the COP2->GPR load-delay check.
|
||||||
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
||||||
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
||||||
|
|||||||
@@ -2388,6 +2388,184 @@ end
|
|||||||
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- GTE control-register alias + RT-diagonal + TR-naming helpers and checks
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Resolve a `gte_cr_<Alias>` ident to its alias-group entry, or nil if the alias
|
||||||
|
--- is in a distinct-slot group (or the alias name is not a known C2 control-register alias).
|
||||||
|
--- Reads `M.GTE_CR_ALIAS_GROUPS` from `duffle.lua`.
|
||||||
|
local function find_alias_pair_for(alias_name, duffle)
|
||||||
|
local groups = (duffle and duffle.GTE_CR_ALIAS_GROUPS) or {}
|
||||||
|
for _, group in ipairs(groups) do
|
||||||
|
for _, name in ipairs(group[2] or {}) do
|
||||||
|
if name == alias_name then return group end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- True iff `c` (a TokClass entry) is a CPU→COP2 control-register transfer
|
||||||
|
-- (`gte_mv_to_ctrl_r` / `gte_mv_from_ctrl_r`).
|
||||||
|
local function is_ctrl_r_transfer(c)
|
||||||
|
if c == nil then return false end
|
||||||
|
return c.ident == "gte_mv_to_ctrl_r" or c.ident == "gte_mv_from_ctrl_r"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Resolve a token's source line. The per-token `line` is the body-relative
|
||||||
|
-- line; `atom.line` is the source line of the atom declaration; `line_in_body`
|
||||||
|
-- (atom.paths) maps a body-relative line to its source line. The arithmetic
|
||||||
|
-- `atom.line + line_in_body[tok.rel] - 1` matches the convention used by
|
||||||
|
-- check_abi_handoff and check_control_transfer_delay_slot_use elsewhere.
|
||||||
|
local function atom_body_token_source_line(atom, token, line_in_body)
|
||||||
|
if line_in_body == nil or token == nil or token.rel == nil then
|
||||||
|
return atom.line or 0
|
||||||
|
end
|
||||||
|
local body_line = line_in_body[token.rel]
|
||||||
|
if body_line == nil then return atom.line or 0 end
|
||||||
|
return (atom.line or 0) + body_line - 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check #N: gte_cr_alias_writes
|
||||||
|
-- Fires one warning per atom per alias-group when the atom body touches two
|
||||||
|
-- distinct aliases from the same group. Aliases within a group write to the
|
||||||
|
-- same C2 control-register slot on real silicon; cross-alias writes inside
|
||||||
|
-- one atom body silently clobber each other.
|
||||||
|
--
|
||||||
|
-- Severity: warning. Build continues. The libgte outer-product convention
|
||||||
|
-- uses only RT-row aliases (which are NOT in `M.GTE_CR_ALIAS_GROUPS`), so
|
||||||
|
-- the canonical convention does not trigger this check.
|
||||||
|
local function check_gte_cr_alias_writes(atom, pipe_ctx, findings)
|
||||||
|
local groups = pipe_ctx.gte_cr_alias_groups or {}
|
||||||
|
if not next(groups) then return end
|
||||||
|
|
||||||
|
local tokens = atom.paths and atom.paths.tokens or {}
|
||||||
|
local tc = atom.paths and atom.paths.tok_class or {}
|
||||||
|
local line_in_body = atom.paths and atom.paths.line_in_body
|
||||||
|
if not next(tokens) then return end
|
||||||
|
|
||||||
|
-- Build a per-group set of (alias, source_line) pairs touched in this atom body.
|
||||||
|
-- Walks every token; when the token is a ctrl-r transfer, the alias is at
|
||||||
|
-- position tok_idx + 2 (rt, alias, [imm-or-arg]). The pre-classified
|
||||||
|
-- `tc` table tells us whether the token is a ctrl-r transfer and what its
|
||||||
|
-- source line is.
|
||||||
|
local touched = {}
|
||||||
|
for tok_idx, token in ipairs(tokens) do
|
||||||
|
local c = tc[tok_idx]
|
||||||
|
if is_ctrl_r_transfer(c) and tokens[tok_idx + 2] then
|
||||||
|
local alias = tokens[tok_idx + 2].tok
|
||||||
|
local group = find_alias_pair_for(alias, pipe_ctx.duffle)
|
||||||
|
if group then
|
||||||
|
touched[group[1]] = touched[group[1]] or {}
|
||||||
|
touched[group[1]][#touched[group[1]] + 1] = {
|
||||||
|
alias = alias,
|
||||||
|
line = atom_body_token_source_line(atom, token, line_in_body),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Fire one warning per group touched with 2+ distinct aliases.
|
||||||
|
for slot, hits in pairs(touched) do
|
||||||
|
local seen = {}
|
||||||
|
local distinct = {}
|
||||||
|
for _, h in ipairs(hits) do
|
||||||
|
if not seen[h.alias] then
|
||||||
|
seen[h.alias] = true
|
||||||
|
distinct[#distinct + 1] = h
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #distinct >= 2 then
|
||||||
|
local aliases = {}
|
||||||
|
for _, d in ipairs(distinct) do aliases[#aliases + 1] = d.alias end
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name or "",
|
||||||
|
line = distinct[1].line,
|
||||||
|
check = "gte_cr_alias_writes",
|
||||||
|
kind = "warning",
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' touches %d aliases that share C2[%d]: %s; verify the intent"
|
||||||
|
, atom.name or "", #distinct, slot, table.concat(aliases, ", ")),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check #N+1: rtdiagonal_completeness
|
||||||
|
-- Fires one info per atom body when the bare `gte_cmdw_mvmva` macro is used.
|
||||||
|
-- The bare macro encodes only the cmd field; the canonical libgte-2-pass
|
||||||
|
-- shape uses `gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012` (gte.h:430).
|
||||||
|
--
|
||||||
|
-- Severity: info by default. Escalates to warning when
|
||||||
|
-- `GTE_RT_DIAGONAL_STRICT=1` env var is set (CI / production builds).
|
||||||
|
--
|
||||||
|
-- The bare macro IS the right call for the canonical libgte outer-product
|
||||||
|
-- convention, so this is an opt-out hint rather than a hard warning.
|
||||||
|
local function check_rtdiagonal_completeness(atom, _pipe_ctx, findings)
|
||||||
|
local tokens = atom.paths and atom.paths.tokens or {}
|
||||||
|
local tc = atom.paths and atom.paths.tok_class or {}
|
||||||
|
local line_in_body = atom.paths and atom.paths.line_in_body
|
||||||
|
if not next(tokens) then return end
|
||||||
|
local strict = os.getenv("GTE_RT_DIAGONAL_STRICT") == "1"
|
||||||
|
for tok_idx, token in ipairs(tokens) do
|
||||||
|
local c = tc[tok_idx]
|
||||||
|
if c and c.ident == "gte_cmdw_mvmva" then
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name or "",
|
||||||
|
line = atom_body_token_source_line(atom, token, line_in_body),
|
||||||
|
check = "rtdiagonal_completeness",
|
||||||
|
kind = strict and "warning" or "info",
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' uses the bare gte_cmdw_mvmva macro; "
|
||||||
|
.. "the canonical libgte-2-pass shape is gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012 "
|
||||||
|
.. "(gte.h:430). The bare macro does not encode RT23/RT31/RT32/RT33; "
|
||||||
|
.. "for a full 3x3 matrix, use the dedicated literal or hand-build via enc_gte_*()."
|
||||||
|
, atom.name or ""),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check #N+2: gte_cr_TR_naming
|
||||||
|
-- Fires one info per atom body when a `gte_cr_TR[XYZ]` alias is used.
|
||||||
|
-- Translation-vector registers are the only 3-letter-suffix C2 aliases
|
||||||
|
-- (`TRX/TRY/TRZ`); an agent who reads `TRX` might typo it as `RT_X` or
|
||||||
|
-- `RTX0` and either get a compile error (best case) or a build that
|
||||||
|
-- links but routes the `ctc2` write to the wrong C2 slot.
|
||||||
|
--
|
||||||
|
-- Severity: info. The convention is correct; this is a documentation-pointer check.
|
||||||
|
local function check_gte_cr_TR_naming(atom, _pipe_ctx, findings)
|
||||||
|
local tokens = atom.paths and atom.paths.tokens or {}
|
||||||
|
local tc = atom.paths and atom.paths.tok_class or {}
|
||||||
|
local line_in_body = atom.paths and atom.paths.line_in_body
|
||||||
|
if not next(tokens) then return end
|
||||||
|
local touched = false
|
||||||
|
local first_line = 0
|
||||||
|
for tok_idx, token in ipairs(tokens) do
|
||||||
|
local c = tc[tok_idx]
|
||||||
|
if c and c.ident and c.ident:match("^gte_cr_TR[XYZ]$") then
|
||||||
|
touched = true
|
||||||
|
if first_line == 0 then
|
||||||
|
first_line = atom_body_token_source_line(atom, token, line_in_body)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if touched then
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name or "",
|
||||||
|
line = first_line,
|
||||||
|
check = "gte_cr_TR_naming",
|
||||||
|
kind = "info",
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' uses gte_cr_TR[XYZ]; translation-vector registers are the only "
|
||||||
|
.. "3-letter-suffix C2 aliases (TRX/TRY/TRZ). See docs/gte_reference.md §"
|
||||||
|
.. "\"The `gte_cmdw_mvmva_c11_pass2_exact` literal\" for the libgte outer-product "
|
||||||
|
.. "convention that uses these names."
|
||||||
|
, atom.name or ""),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
|
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
@@ -2414,6 +2592,9 @@ local CHECK_RULES = {
|
|||||||
{ name = "abi_handoff", per_atom = check_abi_handoff },
|
{ name = "abi_handoff", per_atom = check_abi_handoff },
|
||||||
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
|
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
|
||||||
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
|
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
|
||||||
|
{ name = "gte_cr_alias_writes", per_atom = check_gte_cr_alias_writes },
|
||||||
|
{ name = "rtdiagonal_completeness", per_atom = check_rtdiagonal_completeness },
|
||||||
|
{ name = "gte_cr_TR_naming", per_atom = check_gte_cr_TR_naming },
|
||||||
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
|
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
|
||||||
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
|
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
|
||||||
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
|
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
|
||||||
@@ -2451,12 +2632,17 @@ local function build_corpus_pipe_ctx(ctx)
|
|||||||
atoms_by_name = corpus.atoms_by_name or {},
|
atoms_by_name = corpus.atoms_by_name or {},
|
||||||
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original
|
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original
|
||||||
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
|
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
|
||||||
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
|
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
|
||||||
components_by_name = corpus.components or {},
|
components_by_name = corpus.components or {},
|
||||||
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
||||||
atom_infos_list = corpus.atom_infos or {},
|
atom_infos_list = corpus.atom_infos or {},
|
||||||
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
||||||
collisions = corpus.collisions or {},
|
collisions = corpus.collisions or {},
|
||||||
|
-- GTE control-register alias groups (from `duffle.GTE_CR_ALIAS_GROUPS`).
|
||||||
|
-- The three new per_atom checks (gte_cr_alias_writes, rtdiagonal_completeness,
|
||||||
|
-- gte_cr_TR_naming) read from this view. `duffle` is exposed alongside so
|
||||||
|
-- `find_alias_pair_for` can resolve alias → group without a separate registry.
|
||||||
|
gte_cr_alias_groups = duffle.GTE_CR_ALIAS_GROUPS or {},
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
Binary file not shown.
@@ -61,6 +61,119 @@ if (-not $msbuild_exe) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
||||||
|
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
# NuGet restore — required before MSBuild.
|
||||||
|
# pcsx-redux's .vcxproj files use the legacy packages.config style with
|
||||||
|
# hardcoded `<Import Project="..\packages\{id}.{ver}\...">` directives.
|
||||||
|
# MSBuild's `/t:Restore` won't fetch missing packages here (the local
|
||||||
|
# packages\ dir is checked but no package-source lookup happens), and
|
||||||
|
# `dotnet restore` errors on packages.config projects, so we walk every
|
||||||
|
# packages.config, parse out the <package id version/> entries, and pull
|
||||||
|
# any missing .nupkg directly from api.nuget.org's flat container.
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
$path_pcsx_packages = join-path $path_pcsx_redux 'vsprojects\packages'
|
||||||
|
$nuget_flat_container = 'https://api.nuget.org/v3-flatcontainer'
|
||||||
|
|
||||||
|
# Collect required (id, version) pairs from every packages.config.
|
||||||
|
$required_packages = @{}
|
||||||
|
Get-ChildItem -Path (join-path $path_pcsx_redux 'vsprojects') -Filter 'packages.config' -Recurse -ErrorAction SilentlyContinue |
|
||||||
|
ForEach-Object {
|
||||||
|
[xml]$xml = Get-Content -LiteralPath $_.FullName -Raw
|
||||||
|
foreach ($pkg in $xml.packages.package) {
|
||||||
|
$key = '{0}|{1}' -f $pkg.id, $pkg.version
|
||||||
|
$required_packages[$key] = @{ id = $pkg.id; version = $pkg.version }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Ensure the packages root exists.
|
||||||
|
if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
|
||||||
|
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
|
||||||
|
}
|
||||||
|
|
||||||
|
# Download anything missing. Skip the package entirely if its dir already has
|
||||||
|
# any contents (the legacy packages.config style means the targets file
|
||||||
|
# location varies per package — `luajit.native` puts it at build/native/,
|
||||||
|
# `glfw` puts it elsewhere — so we can't probe a specific path; just check
|
||||||
|
# whether the dir is non-empty).
|
||||||
|
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||||
|
foreach ($pkg in $required_packages.Values) {
|
||||||
|
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
|
||||||
|
if ((Test-Path -LiteralPath $pkgDir) -and `
|
||||||
|
(@(Get-ChildItem -LiteralPath $pkgDir -Recurse -ErrorAction SilentlyContinue).Count -gt 0)) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
$url = '{0}/{1}/{2}/{1}.{2}.nupkg' -f $nuget_flat_container, $pkg.id, $pkg.version
|
||||||
|
$nupkg = Join-Path $pkgDir ('{0}.{1}.nupkg' -f $pkg.id, $pkg.version)
|
||||||
|
New-Item -ItemType Directory -Path $pkgDir -Force | Out-Null
|
||||||
|
Write-Host "Fetching NuGet package: $($pkg.id) $($pkg.version)"
|
||||||
|
try {
|
||||||
|
Invoke-WebRequest -Uri $url -OutFile $nupkg -UseBasicParsing -ErrorAction Stop
|
||||||
|
[System.IO.Compression.ZipFile]::ExtractToDirectory($nupkg, $pkgDir)
|
||||||
|
Remove-Item -LiteralPath $nupkg -Force
|
||||||
|
} catch {
|
||||||
|
$msg = $_.Exception.Message
|
||||||
|
if ($msg -match '404') {
|
||||||
|
Write-Host " Not on nuget.org (vendored?) — skipping $url"
|
||||||
|
} else {
|
||||||
|
Write-Warning "Failed to fetch $url — $msg"
|
||||||
|
}
|
||||||
|
if (Test-Path -LiteralPath $nupkg) { Remove-Item -LiteralPath $nupkg -Force }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into
|
||||||
|
# luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick. The raw string
|
||||||
|
# literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or
|
||||||
|
# MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384,
|
||||||
|
# minus 5 bytes for the `-- lualoader, ` prefix). If the upstream file
|
||||||
|
# grows past that, trim it: remove license header, trailing whitespace,
|
||||||
|
# blank separators, inline comments, and shrink 4-space indent to 2-space.
|
||||||
|
# Idempotent — only writes when the raw string exceeds the limit.
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
|
||||||
|
if (Test-Path -LiteralPath $path_isoffi) {
|
||||||
|
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
|
||||||
|
$startMarker = $content.IndexOf('R"EOF(--')
|
||||||
|
$endMarker = $content.IndexOf('-- )EOF"')
|
||||||
|
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) {
|
||||||
|
$endMarker - ($startMarker + 8)
|
||||||
|
} else { -1 }
|
||||||
|
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
|
||||||
|
if ($literalLen -gt 16379) {
|
||||||
|
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
|
||||||
|
$lines = $content -split "`n"
|
||||||
|
$markerIdx = -1
|
||||||
|
for ($i = 0; $i -lt $lines.Length; $i++) {
|
||||||
|
if ($lines[$i] -match '^-- \)EOF"') { $markerIdx = $i; break }
|
||||||
|
}
|
||||||
|
$newLines = @()
|
||||||
|
for ($i = 0; $i -lt $lines.Length; $i++) {
|
||||||
|
$lineNum = $i + 1
|
||||||
|
$line = $lines[$i]
|
||||||
|
# Keep the first line and the EOF-marker line untouched.
|
||||||
|
if ($i -eq 0 -or $i -eq $markerIdx) { $newLines += $line; continue }
|
||||||
|
# Drop the GPL license header (lines 2-17).
|
||||||
|
if ($lineNum -ge 2 -and $lineNum -le 17) { continue }
|
||||||
|
# Drop blank separator lines.
|
||||||
|
if ($line -match '^\s*$') { continue }
|
||||||
|
# Drop trailing whitespace.
|
||||||
|
$line = $line -replace '\s+$', ''
|
||||||
|
# Drop inline comments (anything from `--` to end of line).
|
||||||
|
$line = $line -replace '\s*--.*$', ''
|
||||||
|
# Shrink 4-space indent to 2-space.
|
||||||
|
$line = $line -replace '^( )', ' '
|
||||||
|
if ($line -match '^\s*$') { continue }
|
||||||
|
$newLines += $line
|
||||||
|
}
|
||||||
|
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
|
||||||
|
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) `
|
||||||
|
-replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
|
||||||
|
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
||||||
|
|
||||||
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
||||||
@@ -117,6 +230,17 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
|||||||
# ════════════════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
||||||
|
|
||||||
|
# Wipe stale *.dep files across src\mips. These cache absolute paths to the
|
||||||
|
# GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
|
||||||
|
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
|
||||||
|
# `make clean` in openbios only clears its own dir — subdirs like
|
||||||
|
# common/crt0/, modplayer/, and shell/ keep their stale .dep files. Easier to
|
||||||
|
# just delete the lot before each build than to teach every Makefile about
|
||||||
|
# deepclean recursion.
|
||||||
|
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
|
||||||
|
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
|
||||||
|
|
||||||
push-location $path_openbios
|
push-location $path_openbios
|
||||||
& make clean
|
& make clean
|
||||||
& make
|
& make
|
||||||
|
|||||||
Reference in New Issue
Block a user