From 3f3b691ac0fbeb57218c17c212386cd63fa7e520 Mon Sep 17 00:00:00 2001 From: Ed_ Date: Tue, 11 Aug 2026 11:25:54 -0400 Subject: [PATCH] Making a proper distinction between atom arenas and atom builders. --- code/duffle/gen/offsets.h | 2 +- code/duffle/gp.atom.c | 12 +-- code/duffle/gte.atom.c | 74 ++++++++------- code/duffle/lottes_tape.h | 54 ++++++----- code/duffle/math.atom.c | 12 +-- code/duffle/memory.h | 10 ++- code/duffle/pad.atom.c | 8 +- code/hello_camera/hello_camera.atom.c | 37 ++++---- code/hello_camera/hello_camera.c | 124 +++++++++++--------------- 9 files changed, 164 insertions(+), 169 deletions(-) diff --git a/code/duffle/gen/offsets.h b/code/duffle/gen/offsets.h index 32af6aa..f6536b2 100644 --- a/code/duffle/gen/offsets.h +++ b/code/duffle/gen/offsets.h @@ -25,7 +25,7 @@ #pragma region duffle -// --- atom: normalize_v3s4 (62 words) --- +// --- atom: normalize_v3s4 (63 words) --- #define _atom_offset_srav_path_aligned_done 6 #define _atom_offset_aligned_done_srav_path 1 diff --git a/code/duffle/gp.atom.c b/code/duffle/gp.atom.c index 80f7150..7c38604 100644 --- a/code/duffle/gp.atom.c +++ b/code/duffle/gp.atom.c @@ -8,30 +8,30 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c); #pragma region MACs (Mips Atom Components) -FI_ Slice_MipsCode ac_gcmd_push(MipsAtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port) +FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port) MipsAtomComp_Proc_(ac_gcmd_push, ab, { load_upper_i(reg_transfer, cmd >> 16), or_i_self( reg_transfer, cmd & 0xFFFF), store_word( reg_transfer, reg_base, port), }) -FI_ Slice_MipsCode ac_store_rgb8(MipsAtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, { +FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, ab, { store_byte(rr, base, offset + O_(RGB8,r)), store_byte(rg, base, offset + O_(RGB8,g)), store_byte(rb, base, offset + O_(RGB8,b)), }) -FI_ Slice_MipsCode ac_pack_color_word(MipsAtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b) +FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b) atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, ab, { load_upper_i(R_AT, (cmd) << 8 | (b)), or_i_self( R_AT, ((g) << 8) | (r)), store_word( R_AT, r_base, (off)), }) -FI_ Slice_MipsCode ac_format_f3_color(MipsAtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b) +FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b) atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) }) -FI_ Slice_MipsCode ac_format_g4_color(MipsAtomBuilder_R ab, U4 r_prim_cursor, +FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor, U1 r0, U1 g0, U1 b0, U1 r1, U1 g1, U1 b1, U1 r2, U1 g2, U1 b2, @@ -44,7 +44,7 @@ MipsAtomComp_Proc_(ac_format_g4_color, ab, { }) /* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */ -I_ Slice_MipsCode ac_insert_ot_tag(MipsAtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, { +I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ac_insert_ot_tag, ab, { shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1) add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ] load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head diff --git a/code/duffle/gte.atom.c b/code/duffle/gte.atom.c index 3b3cbce..fb75e02 100644 --- a/code/duffle/gte.atom.c +++ b/code/duffle/gte.atom.c @@ -11,7 +11,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c); #pragma region MACs (Mips Atom Components) /* Words: 3; Loads 3 S2 indices from the face array */ -FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, { +FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2) atom_dbg_skip MipsAtomComp_Proc_(ac_load_tri_indices, ab, { load_half_u(r_i0, r_face_cusor, 0 * S_(S2)), load_half_u(r_i1, r_face_cusor, 1 * S_(S2)), load_half_u(r_i2, r_face_cusor, 2 * S_(S2)), @@ -19,14 +19,14 @@ FI_ Slice_MipsCode ac_load_tri_indices(MipsAtomBuilder_R ab, U4 r_face_cusor, U4 /* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3. * PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */ -FI_ Slice_MipsCode ac_gte_store_f3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, { +FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_f3, ab, { gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)), gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)), gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)), }) /* Words: 18; Translates indices to vertex addresses and pushes them to GTE */ -I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, { +I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_load_tri_verts, ab, { shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0), shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1), shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2), @@ -37,7 +37,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(MipsAtomBuilder_R ab, U4 r_vert_base, U4 * PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). * MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3 * (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */ -FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, { +FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p012, ab, { gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)), gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)), gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)), @@ -47,13 +47,13 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(MipsAtomBuilder_R ab, U4 r_primitive_cur * PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2; * SXY0 still holds v0.screen from the earlier RTPT. */ -FI_ Slice_MipsCode ac_gte_store_g4_p3(MipsAtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) }) +FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_store_g4_p3, ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) }) /* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ─── * Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. * Stage 2 of normalize consumes these directly. * Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */ -FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, { +FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_sqr_v3, ab, { gte_mv_to_data_r(r_sx, C2_IR1), gte_mv_to_data_r(r_sy, C2_IR2), gte_mv_to_data_r(r_sz, C2_IR3), @@ -68,7 +68,7 @@ FI_ Slice_MipsCode ac_gte_sqr_v3(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz * (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output. * Used standalone for "scale vector by scalar". * Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */ -FI_ Slice_MipsCode ac_gte_gpf_scale(MipsAtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, { +FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_recip_est, U4 r_shift, U4 r_dx, U4 r_dy, U4 r_dz) atom_dbg_skip MipsAtomComp_Proc_(ac_gte_gpf_scale, ab, { gte_mv_to_data_r(r_recip_est, C2_IR0), gte_mv_to_data_r(r_sx, C2_IR1), gte_mv_to_data_r(r_sy, C2_IR2), @@ -165,13 +165,13 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = { * * Body uses 9 GPRs (r_src_ptr..r_branch_tmp): * r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets) - * r_tmp : scratch (reserved for misc use) - * r_mac1_scratch : MAC1 result scratch (before sum into r_recip_est) - * r_mac2_scratch : MAC2 result scratch (clobbered to IR1 in stage 4) - * r_recip_est : |v|² sum + shift-input + sqrtbl[index] (the main chain) - * r_lzcr : LZCR value (consumed by stage 3 alignment calc) - * r_shift : final srav amount (consumed by stage 4 shift_aright_var) - * r_branch_tmp : scratch (shift count, branch target, sqrtbl base addr) + * r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4 + * r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3) + * r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra + * r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y + * r_lzcr : |v|² sum (stage 2) → shift count (stage 3) → 1/|v| (stage 4 IR0) + * r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav + * r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr) * * Atom_labels are srav_path / aligned_done * (NOT namespaced — they're internal to this proc; @@ -185,7 +185,7 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = { * Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR. */ /* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */ -I_ void normalize_v3s4_proc(MipsAtomBuilder_R ab, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */ +I_ MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */ , U4 r_src_offset, U4 r_dst_offset /* GPR codes: PARAMETERIZED offsets (caller passes O_ macros) */ , U4 r_src_ptr, U4 r_dst_ptr, U4 r_tmp /* GPR codes: 3 scratch regs (src/dst computed + tmp) */ , U4 r_mac1_scratch, U4 r_mac2_scratch /* GPR codes: 2 more: MAC1/MAC2 scratch */ @@ -193,21 +193,22 @@ I_ void normalize_v3s4_proc(MipsAtomBuilder_R ab, U4 r_scratch /* GPR code: scra , U4 r_lzcr, U4 r_shift /* GPR codes: lzcr + final srav amount */ , U4 r_branch_tmp /* GPR code: scratch (shift count, branch target, lookup addr) */ ) -MipsAtom_Proc_(normalize_v3s4, ab, { +MipsAtom_Proc_(normalize_v3s4, aa, { add_si(r_src_ptr, r_scratch, r_src_offset), /* r_src_ptr = &src */ add_si(r_dst_ptr, r_scratch, r_dst_offset), /* r_dst_ptr = &dst */ nop, - /* Load src.x/y/z from r_src_ptr (caller-determined address) into r_mac2_scratch/r_recip_est/r_branch_tmp. */ - load_word(r_mac2_scratch, r_src_ptr, O_(V3_S4,x)), + /* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp. + * r_tmp holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */ + load_word(r_tmp, r_src_ptr, O_(V3_S4,x)), load_word(r_recip_est, r_src_ptr, O_(V3_S4,y)), load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)), nop, /* load-delay */ /* Stage 1: mtc2 src → IR1/2/3, SQR fires. */ - gte_mv_to_data_r(r_mac2_scratch, C2_IR1), - gte_mv_to_data_r(r_recip_est, C2_IR2), - gte_mv_to_data_r(r_branch_tmp, C2_IR3), + gte_mv_to_data_r(r_tmp, C2_IR1), + gte_mv_to_data_r(r_recip_est, C2_IR2), + gte_mv_to_data_r(r_branch_tmp, C2_IR3), nop, gte_cmdw_sqr, /* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */ @@ -222,7 +223,9 @@ MipsAtom_Proc_(normalize_v3s4, ab, { gte_mv_from_data_r(r_shift, C2_LZCR), nop, - /* Stage 3: compute srav amount (r_lzcr) + align |v|² to bit 24. */ + /* Stage 3: compute srav amount (r_lzcr) + align |v|² to bit 24. + * IMPORTANT: the sllv/srav below writes the aligned |v|² to r_mac1_scratch (NOT r_lzcr), + * so r_lzcr retains the shift count all the way to the start of stage 4. */ and_i( r_shift, r_shift, -2), li_s( r_lzcr, 31), sub_s( r_lzcr, r_lzcr, r_shift), @@ -231,32 +234,35 @@ MipsAtom_Proc_(normalize_v3s4, ab, { add_si( r_branch_tmp, r_shift, -24), branch_lt_zero(r_branch_tmp, atom_offset(srav_path, aligned_done)), nop, jump_rel(atom_offset(aligned_done, srav_path)), - shift_lleft_var(r_lzcr, r_lzcr, r_branch_tmp), /* when r_branch_tmp < 0 (LZCR < 24): shift r_lzcr left by (24-LZCR) */ + shift_lleft_var(r_mac1_scratch, r_lzcr, r_branch_tmp), /* sllv path: aligned = r_lzcr sll (LZCR-24) — dst=r_mac1_scratch to PRESERVE r_lzcr=shift count */ atom_label(srav_path) li_s( r_branch_tmp, 24), sub_s( r_branch_tmp, r_branch_tmp, r_shift), - shift_aright_var(r_lzcr, r_lzcr, r_branch_tmp), /* when r_branch_tmp >= 0 (LZCR >= 24): shift r_lzcr right by (LZCR-24) */ + shift_aright_var(r_mac1_scratch, r_lzcr, r_branch_tmp), /* srav path: aligned = r_lzcr sra (24-LZCR) — dst=r_mac1_scratch to PRESERVE r_lzcr=shift count */ atom_label(aligned_done) - /* r_lzcr holds |v|² aligned to bit 24. */ - add_si( r_lzcr, r_lzcr, -64), - shift_lleft(r_lzcr, r_lzcr, 1), + /* Save the shift count to r_shift before the next 5 instructions overwrite r_lzcr + * (the sqrtbl lookup loads 1/|v| into r_lzcr, which becomes IR0 in stage 4). */ + or_u(r_shift, r_lzcr, 0), /* r_shift ← shift count (preserved through stage 4) */ + /* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */ + add_si( r_mac1_scratch, r_mac1_scratch, -64), + shift_lleft(r_mac1_scratch, r_mac1_scratch, 1), load_upper_i(r_branch_tmp, u4_hi(& gte_normalize_sqr_tbl)), or_i_self( r_branch_tmp, u4_lo(& gte_normalize_sqr_tbl)), - add_u(r_branch_tmp, r_branch_tmp, r_lzcr), - load_half(r_lzcr, r_branch_tmp, 0), nop, + add_u(r_branch_tmp, r_branch_tmp, r_mac1_scratch), + load_half(r_lzcr, r_branch_tmp, 0), nop, /* r_lzcr = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */ - /* Stage 4: GPF + srav finalize (r_lzcr = srav_amount carried from stage 3). */ + /* Stage 4: GPF + srav finalize (r_shift = shift count, r_lzcr = 1/|v|). */ gte_mv_to_data_r(r_lzcr, C2_IR0), - gte_mv_to_data_r(r_mac2_scratch, C2_IR1), + gte_mv_to_data_r(r_tmp, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ gte_mv_to_data_r(r_recip_est, C2_IR2), gte_mv_to_data_r(r_branch_tmp, C2_IR3), nop2, gte_cmdw_gpf, gte_mv_from_data_r(r_mac2_scratch, C2_MAC1), gte_mv_from_data_r(r_recip_est, C2_MAC2), gte_mv_from_data_r(r_branch_tmp, C2_MAC3), - shift_aright_var(r_mac2_scratch, r_mac2_scratch, r_lzcr), - shift_aright_var(r_recip_est, r_recip_est, r_lzcr), - shift_aright_var(r_branch_tmp, r_branch_tmp, r_lzcr), + shift_aright_var(r_mac2_scratch, r_mac2_scratch, r_shift), /* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */ + shift_aright_var(r_recip_est, r_recip_est, r_shift), + shift_aright_var(r_branch_tmp, r_branch_tmp, r_shift), /* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */ store_word(r_mac2_scratch, r_dst_ptr, O_(V3_S4,x)), diff --git a/code/duffle/lottes_tape.h b/code/duffle/lottes_tape.h index b40a356..34faa24 100644 --- a/code/duffle/lottes_tape.h +++ b/code/duffle/lottes_tape.h @@ -110,7 +110,7 @@ typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that // FI_ void ac_X(args) MipsAtomComp_Proc_(ac_X, { body }) // expands to: // FI_ void ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return ac_X; } -#define MipsAtom_Proc_(sym, abuilder, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(abuilder, slice_from_array(MipsCode, sym)); } +#define MipsAtom_Proc_(sym, aa, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, sym)); } // Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names). // MipsAtomComp_(ac_X) { body } @@ -128,7 +128,7 @@ typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that // The body must NOT include mac_yield() (the parent atom yields). // Inline-only callers (the generated `mac_` aliases) skip this arg via metaprogram filtering; // escape callers (ac_ invoked as a function) pass a long-lived builder. -#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_unroll(ab, slice_from_array(MipsCode, sym)); } +#define MipsAtomComp_Proc_(sym, ab, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, sym)); } /* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content. Files containing only atoms and atom components. @@ -231,44 +231,56 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) { }; #pragma endregion Macro Atom Components -#pragma region Mips Atom Builder +#pragma region Atom Builder // This helps with runtime procedural authoring of mips atoms. typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; }; // FArena Related -typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; }; +typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; }; // Whatever the builder is writting to should most likely coresspond // to something that can fit within instruction cache? -FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode code) { - /* code.len is in ELEMENTS (per slice_from_array convention); ab->used is also in elements - * (the init uses `ab->used * sizeof(U4)` for byte offset arithmetic — sizeof(U4)==4==sizeof(MipsCode)). - * mem_copy needs BYTES, so we use S_slice(code) for the length. */ +// Usual way to resolve an atom after the bulder is done. +#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start) + +FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) { assert(ab->capacity - ab->used - code.len); - U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot (arena accumulation) */ - mem_copy(u4_(dest), u4_(code.ptr), S_slice(code)); + U4 dest = ab->start + ab->used * S_(MipsCode); + mem_copy(dest, u4_(code.ptr), S_slice(code)); mem_bump(ab->start, ab->capacity, & ab->used, code.len); } -#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac)) +#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac)) // When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc). -FI_ void atombuilder_end(MipsAtomBuilder_R ab) { - /* ac_yield is a MipsCode[] of 4 elements; S_(ac_yield)=bytes, array_len(ac_yield)=elements. - * ab->used is in elements, so mem_bump needs element count. */ - U4* dest = (U4*)ab->start + ab->used; /* write at next-available slot */ - mem_copy(u4_(dest), u4_(ac_yield), S_(ac_yield)); - mem_bump(ab->start, ab->capacity, & ab->used, array_len(ac_yield)); -} - -#define mipsatom_from_builder(ab) C_(MipsAtom*, (ab).start) +FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); } // tb_emit_builder(tb, ab) — emit the builder's atom into the tape and advance tb->used. // Thin wrapper around tb_emit(tb, mipsatom_from_builder(ab[0])). // Equivalent to tb_emit(tb, code_) for runtime-built atoms. -FI_ void tb_emit_builder(TapeBuilder_R tb, MipsAtomBuilder_R ab) { tb_emit(tb, mipsatom_from_builder(ab[0])); } +FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); } #pragma endregion Mips Atom Builder +#pragma region Atom Arena +typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; }; + +#define atomarena_unused_start(ab) ((ab).start + (ab).used * S_(MipsCode)) +FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr); + arena->start = u4_(mem.ptr); + arena->capacity = mem.len; + arena->used = 0; +} +FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; } +FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) { + assert(aa->capacity - aa->used - code.len); + U4 dest = atomarena_unused_start(aa[0]); + mem_copy(dest, u4_(code.ptr), S_slice(code)); + mem_bump(aa->start, aa->capacity, & aa->used, code.len); + return C_(MipsAtom*, dest); +} +FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; } +#pragma region Atom Arena + #pragma region Mips Atom Procs #pragma endregion Mips Atom Procs diff --git a/code/duffle/math.atom.c b/code/duffle/math.atom.c index 6c0c4ea..e9eee41 100644 --- a/code/duffle/math.atom.c +++ b/code/duffle/math.atom.c @@ -9,35 +9,35 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c); #pragma region MACs (Mips Atom Component) -FI_ Slice_MipsCode ac_load_v2s2(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, { +FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, ab, { load_half( rs_x, r_base, O_(V3_S2,x)), load_half( rs_y, r_base, O_(V3_S2,y)), }) -FI_ Slice_MipsCode ac_store_v2s2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, { +FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, ab, { store_half(rt_x, base, offset + O_(V2_S2,x)), store_half(rt_y, base, offset + O_(V2_S2,y)), }) -FI_ Slice_MipsCode ac_load_v3s4(MipsAtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, { +FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v3s4, ab, { load_word( rs_x, r_base, O_(V3_S4,x)), load_word( rs_y, r_base, O_(V3_S4,y)), load_word( rs_z, r_base, O_(V3_S4,z)), }) -FI_ Slice_MipsCode ac_store_v3s4(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, { +FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v3s4, ab, { store_word(rt_x, base, offset + O_(V3_S4,x)), store_word(rt_y, base, offset + O_(V3_S4,y)), store_word(rt_z, base, offset + O_(V3_S4,z)), }) -FI_ Slice_MipsCode ac_sub_v3s4(MipsAtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, { +FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ac_sub_v3s4, ab, { sub_s(rds_x, rds_x, rt_x), sub_s(rds_y, rds_y, rt_y), sub_s(rds_z, rds_z, rt_z), }) -FI_ Slice_MipsCode ac_store_rects2(MipsAtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, { +FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, ab, { store_half(rt_x, base, offset + O_(Rect_S2,x)), store_half(rt_y, base, offset + O_(Rect_S2,y)), store_half(rt_width, base, offset + O_(Rect_S2,width)), diff --git a/code/duffle/memory.h b/code/duffle/memory.h index cb3be73..14815d1 100644 --- a/code/duffle/memory.h +++ b/code/duffle/memory.h @@ -73,9 +73,9 @@ typedef Slice_(B1); #define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter) #define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) } -#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) / S_(type) } /* .len in elements (matches S_slice/slice_arg_from_array convention) */ +#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) / S_(type) } -FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), S_slice(s)); } +FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); } #define slice_zero(s) slice_zero_(slice_to_ut(s)) FI_ void slice_copy_(Slice dest, Slice src) { @@ -89,6 +89,7 @@ FI_ void slice_copy_(Slice dest, Slice src) { slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \ } while(0) +typedef Slice_(U1); typedef Slice_(U4); #pragma endregion Slice @@ -99,7 +100,7 @@ typedef Opt_(farena) { U4 alignment, type_width; }; typedef Struct_(FArena) { U4 start, capacity, used; }; FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr); arena->start = u4_(mem.ptr); - arena->capacity = S_slice(mem); /* FArena.used is in BYTES; capacity must be bytes too */ + arena->capacity = mem.len; arena->used = 0; } FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; } @@ -116,7 +117,8 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) { U4 end = arena->start + arena->used; assert_bounds(save_point, arena->start, end); arena->used -= save_point - arena->start; } -FI_ U4 farena_save(FArena arena) { return arena.used; } +FI_ U4 farena_save(FArena arena) { return arena.used; } +FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; } #define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__)) #define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr) #define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) } diff --git a/code/duffle/pad.atom.c b/code/duffle/pad.atom.c index 4167273..4df64d1 100644 --- a/code/duffle/pad.atom.c +++ b/code/duffle/pad.atom.c @@ -11,18 +11,18 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c); #pragma region MACs (Mips Atom Components) -FI_ Slice_MipsCode ac_pad_set_centered_axes(MipsAtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, { +FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, U4 r_state, U4 r_scratch) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_centered_axes, ab, { load_upper_i(r_scratch, (PadAxis_Centered_Word >> 16) & 0xFFFF), or_i_self( r_scratch, PadAxis_Centered_Word & 0xFFFF), store_word( r_scratch, r_state, O_(PadState,axes)), }) -FI_ Slice_MipsCode ac_pad_set_id_byte(MipsAtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, { +FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, U1 r_state, U1 r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_id_byte, ab, { add_ui( r_id, R_0, id_value), store_byte(r_id, r_state, O_(PadState,id)), }) -FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, { +FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_set_status, ab, { add_ui( r_tmp, R_0, pad_status), store_word(r_tmp, r_state, O_(PadState,status)), }) @@ -30,7 +30,7 @@ FI_ Slice_MipsCode ac_pad_set_status(MipsAtomBuilder_R ab, U4 r_tmp, U1 r_state, /* Invert r_buttons (active-low → active-high) and store to PadState.buttons. * r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of * the preceding load_half_u with an instruction that doesn't read r_buttons). */ -FI_ Slice_MipsCode ac_pad_store_inverted_buttons(MipsAtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, { +FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ac_pad_store_inverted_buttons, ab, { nor_u( r_buttons, r_buttons, R_0), store_half( r_buttons, r_pad_state, O_(PadState, buttons)), }) diff --git a/code/hello_camera/hello_camera.atom.c b/code/hello_camera/hello_camera.atom.c index 50b9ee4..31591d0 100644 --- a/code/hello_camera/hello_camera.atom.c +++ b/code/hello_camera/hello_camera.atom.c @@ -25,7 +25,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c); #pragma region MACs (Mips Atom components) -FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) +FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) MipsAtomComp_Proc_(ac_put_disp_env, ab, { // Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)). // Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR @@ -36,7 +36,7 @@ MipsAtomComp_Proc_(ac_put_disp_env, ab, { mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), }) -FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) +FI_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port) MipsAtomComp_Proc_(ac_put_draw_env, ab, { /* * ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings. @@ -131,11 +131,6 @@ typedef Struct_(Binds_ResolveLookAt) { V3_S4* up_in; }; -/* Per-atom bind-pop structs for the resolve_look_at bundle. */ -typedef Struct_(Binds_ResolveLookAtScratch) { - U4 scratch_base; /* U4 (scratch base address — populated by helper with u4_(smem.scratchpad)) */ -}; - /* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's * scratchpad slots (PS1 hardware scratchpad at 0x1F800000). * @@ -195,6 +190,7 @@ typedef Struct_(Binds_ResolveLookAtSub) { U4 target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */ U4 eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */ U4 up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */ + U4 scratchpad; }; /* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. @@ -202,15 +198,13 @@ typedef Struct_(Binds_ResolveLookAtSub) { * r_target_ptr : P3_S4* (C-side struct; atom 0 reads target.x/y/z directly) * r_eye_ptr : P3_S4* (C-side struct; staged into scratchpad at +96/+100/+104) * r_up_in_ptr : V3_S4* (C-side struct; staged into scratchpad at +128/+132/+136) - * Wave-context output: * r_scratch : R_ResolveScratch (R_T4) — scratch base, read by atoms 1-6 * * Bind-pop layout: - * Binds_ResolveLookAtSub = 12 bytes (target + eye + up_in ptrs) - * Binds_ResolveLookAtScratch = 4 bytes (scratch_base) + * Binds_ResolveLookAtSub * Staging work: - * * Stage eye.x/y/z → scratch+96/+100/+104 (for atom 6's translation column) - * * Stage up_in.x/y/z → scratch+128/+132/+136 (for atom 2's outer-product operand) + * * Stage eye.x/y/z → scratch (for atom 6's translation column) + * * Stage up_in.x/y/z → scratch (for atom 2's outer-product operand) * * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1) * * GPR codes (assigned by resolve_look_at_init): @@ -227,17 +221,16 @@ typedef Struct_(Binds_ResolveLookAtSub) { * * Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs. */ -I_ void resolve_look_at__input_and_sub_proc(MipsAtomBuilder_R ab, U4 r_scratch +I_ MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa, U4 r_scratch , U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr , U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3 -) MipsAtom_Proc_(resolve_look_at__input_and_sub, ab, { +) MipsAtom_Proc_(resolve_look_at__input_and_sub, aa, { /* Pop the 3 C-side pointers + scratch_base from the tape. */ load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)), load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)), load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)), + load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)), add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)), - load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtScratch,scratch_base)), - add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtScratch)), /* Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation * column). Reuse r_tmp0/r_tmp1/r_tmp2. Offsets via O_(ResolveLookAtScratch,*). */ @@ -295,11 +288,11 @@ I_ void resolve_look_at__input_and_sub_proc(MipsAtomBuilder_R ab, U4 r_scratch */ /* Atom 2: cross uz × up_in → right. */ -I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r_scratch +I_ MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch , U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */ , U4 r_d /* load b.x */ , U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */ -) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, ab, { +) MipsAtom_Proc_(resolve_look_at__cross_uz_up_in_to_right, aa, { /* Compute the three scratch pointers from r_scratch. */ add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */ add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */ @@ -346,11 +339,11 @@ I_ void resolve_look_at__cross_uz_up_in_to_right_proc(MipsAtomBuilder_R ab, U4 r }) /* Atom 4: cross uz × ux → up. */ -I_ void resolve_look_at__cross_uz_ux_to_up_proc(MipsAtomBuilder_R ab, U4 r_scratch +I_ MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch , U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */ , U4 r_d /* load b.x */ , U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */ -) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, ab, { +) MipsAtom_Proc_(resolve_look_at__cross_uz_ux_to_up, aa, { /* Compute the three scratch pointers from r_scratch. */ add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */ add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */ @@ -415,12 +408,12 @@ typedef Struct_(Binds_ResolveLookAtPopAndTrans) { * MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye). * Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs. */ -I_ void resolve_look_at__populate_and_translate_proc(MipsAtomBuilder_R ab +I_ MipsAtom* resolve_look_at__populate_and_translate_proc(AtomArena_R aa , U4 r_look_at , U4 r_scratch , U4 r_pux, U4 r_puy, U4 r_puz, U4 r_peye /* 4 dedicated pointer regs */ , U4 r_tmp0, U4 r_tmp1, U4 r_tmp2 /* 3 atom-local scratch regs */ -) MipsAtom_Proc_(resolve_look_at__populate_and_translate, ab, { +) MipsAtom_Proc_(resolve_look_at__populate_and_translate, aa, { /* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */ load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)), add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)), diff --git a/code/hello_camera/hello_camera.c b/code/hello_camera/hello_camera.c index 328388d..e0aa325 100644 --- a/code/hello_camera/hello_camera.c +++ b/code/hello_camera/hello_camera.c @@ -52,6 +52,11 @@ #include "hello_camera.atom.c" #pragma endregion Hello Joypad TUs +enum { + Scratchpad_Loc = 0x1F800000, +}; +#define C_scratch(type) C_(type, Scratchpad_Loc) + enum { Scratchpad_Len = 1024, MemTape_Len = 512, @@ -76,18 +81,11 @@ typedef Struct_(SMemory) { PadBiosRaw pad_raw[2]; PadState pad[2]; + // TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address. U4_V scratchpad; // d-cache - /* resolve_look_at bundle: pre-built atom arena + atom-refs. - * (Task 12.5 fix: moved from file-scope globals to smem fields. - * Task 12.7 fix: dropped the ResolveLookAtScratch struct-as-view; the - * C-side helper uses `& smem.scratchpad[N]` at hardcoded offsets directly. - * Task 12.8 fix: chain atoms use r_scratch + offset internally; no C-side magic offsets anywhere. - * Task 12.11 fix: ResolveLookAtScratch offset schema moved to hello_camera.atom.c — gte.atom.c - * is the GENERIC GTE primitives file and must not know about the resolve_look_at bundle's scratch layout.) */ - U4 resolve_look_at_arena[ResolveLookAtArena_Words]; /* ~2 KB; bumped from 420 per Task 4 subagent */ - MipsAtom* resolve_look_at_atom_addrs[7]; - MipsAtomBuilder resolve_look_at_ab_static; + U4 resolve_look_at_mem[ResolveLookAtArena_Words]; + MipsAtom* resolve_look_at_atom_addrs[7]; }; global SMemory smem; extern SMemory smem; @@ -160,37 +158,33 @@ resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) */ internal void resolve_look_at_init(void) { /* Wrap the static arena in a MipsAtomBuilder. */ - MipsAtomBuilder_R ab = & smem.resolve_look_at_ab_static; - ab->start = u4_(smem.resolve_look_at_arena); - ab->capacity = ResolveLookAtArena_Words; - ab->used = 0; + AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem)); /* Atom 0: resolve_look_at__input_and_sub — stages eye/up_in into scratchpad, * computes fwd = target - eye; binds R_ResolveScratch (R_T4) as the wave-context carrier for atoms 1-6. * The body hardcodes R_AT and R_V0 as eye.y/eye.z temps (the existing sub_u(eye.x, eye.y, eye.z) chain from the prior Task 12.7 design). */ - smem.resolve_look_at_atom_addrs[0] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - resolve_look_at__input_and_sub_proc(ab, R_ResolveScratch, + smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab, R_ResolveScratch, R_T0, /* r_target_ptr (popped from tape) */ R_T1, /* r_eye_ptr (popped from tape) */ R_T2, /* r_up_in_ptr (popped from tape) */ R_T3, R_T5, R_T6, R_T7); /* r_tmp<0-3> */ - /* Atom 1: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+0=fwd, dst=scratch+16=uz. + /* Atom 1: normalize_v3s4_proc * The proc takes r_src_offset + r_dst_offset as U4 PARAMETERS — we pass the O_(...) macros here (evaluating to numeric literals 0 and 16). - * The 4-stage body is identical across the 3 call sites (atoms 1, 3, 5); only the offset args differ. + * Body is identical across the 3 call sites (atoms 1, 3, 5); only the offset args differ. * GPR pool: r_scratch (R_T4 carrier) + 9 body GPRs = 10. * r_src_ptr (R_T0) : src ptr * r_dst_ptr (R_T1) : dst ptr - * r_tmp (R_T2) : unused (reserved for symmetry) - * r_mac1_scratch (R_T3) : MAC1 scratch - * r_mac2_scratch (R_T5) : src.x → result.x (carries through stages 1-2) + * r_tmp (R_T2) : src.x PRESERVED (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4 + * r_mac1_scratch (R_T3) : MAC1 scratch + aligned |v|² in stage 3 + * r_mac2_scratch (R_T5) : MAC2 scratch → result.x after stage 4 sra * r_recip_est (R_T6) : src.y → result.y - * r_lzcr (R_T7) : |v|² accumulator + srav amount (single reg) - * r_shift (R_V0) : LZCR (saved across stages 3-4) + * r_lzcr (R_T7) : |v|² accumulator + shift count + 1/|v| (overwritten across stages 2-4) + * r_shift (R_V0) : shift count (saved in stage 3) → sra amount in stage 4 * r_branch_tmp (R_V1) : src.z → result.z (reused after stage 1) */ - smem.resolve_look_at_atom_addrs[1] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - normalize_v3s4_proc(ab, R_ResolveScratch, /* r_scratch (wave-context carrier) */ + ab.start = ab.start + ab.used; + smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab, R_ResolveScratch, O_(ResolveLookAtScratch, fwd), /* r_src_offset = 0 */ O_(ResolveLookAtScratch, uz), /* r_dst_offset = 16 */ R_T0, R_T1, R_T2, /* r_src_ptr, r_dst_ptr, r_tmp */ @@ -201,19 +195,17 @@ internal void resolve_look_at_init(void) { R_V0, /* r_shift */ R_V1); /* r_branch_tmp */ - /* Atom 2: resolve_look_at__cross_uz_up_in_to_right — a=scratch+16, b=scratch+128, + /* Atom 2: resolve_look_at__cross_uz_up_in_to_right * out=scratch+32 (HARDCODED in body). GPR pool: r_scratch + 7 body + R_AT + R_V0 = 10. */ - smem.resolve_look_at_atom_addrs[2] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - resolve_look_at__cross_uz_up_in_to_right_proc(ab, R_ResolveScratch, /* r_scratch (wave-context carrier; src/dst base) */ + smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab, R_ResolveScratch, R_T0, R_T1, R_T2, /* r_a, r_b, r_c (a.x/y/z → out.x/y/z) */ R_T3, /* r_d (b.x) */ R_T5, /* r_f (out ptr = scratch+32) */ R_T6, /* r_g (a ptr = scratch+16) */ R_T7); /* r_h (b ptr = scratch+128) */ - /* Atom 3: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+32=right, dst=scratch+48=ux. */ - smem.resolve_look_at_atom_addrs[3] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - normalize_v3s4_proc(ab, R_ResolveScratch, + /* Atom 3: normalize_v3s4_proc. */ + smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab, R_ResolveScratch, O_(ResolveLookAtScratch, right), /* r_src_offset = 32 */ O_(ResolveLookAtScratch, ux), /* r_dst_offset = 48 */ R_T0, R_T1, R_T2, @@ -225,8 +217,7 @@ internal void resolve_look_at_init(void) { R_V1); /* Atom 4: resolve_look_at__cross_uz_ux_to_up — a=scratch+16, b=scratch+48, out=scratch+64 (HARDCODED). */ - smem.resolve_look_at_atom_addrs[4] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - resolve_look_at__cross_uz_ux_to_up_proc(ab, R_ResolveScratch, + smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab, R_ResolveScratch, R_T0, R_T1, R_T2, R_T3, R_T5, /* r_f (out ptr = scratch+64) */ @@ -234,8 +225,7 @@ internal void resolve_look_at_init(void) { R_T7); /* r_h (b ptr = scratch+48) */ /* Atom 5: normalize_v3s4_proc (generic, from gte.atom.c) — src=scratch+64=up, dst=scratch+80=uy. */ - smem.resolve_look_at_atom_addrs[5] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - normalize_v3s4_proc(ab, R_ResolveScratch, + smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab, R_ResolveScratch, O_(ResolveLookAtScratch, up), /* r_src_offset = 64 */ O_(ResolveLookAtScratch, uy), /* r_dst_offset = 80 */ R_T0, R_T1, R_T2, @@ -248,15 +238,14 @@ internal void resolve_look_at_init(void) { /* Atom 6: resolve_look_at__populate_and_translate — write look_at->m[][] from ux/uy/uz (computed from r_scratch+offset internally), then compute translation column t[] = R * (-eye). GPR pool: r_look_at + r_scratch + 4 ptr regs + 3 tmp regs = 9. */ - smem.resolve_look_at_atom_addrs[6] = (MipsAtom*)u4_v(ab->start + ab->used * sizeof(U4)); - resolve_look_at__populate_and_translate_proc(ab, + smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_and_translate_proc(& ab, R_T0, /* r_look_at (popped from tape; MT3_S2S4*) */ R_ResolveScratch, /* r_scratch (wave-context carrier) */ R_T1, R_T3, R_T5, R_T7, /* r_pux, r_puy, r_puz, r_peye */ R_T2, R_T6, R_V0); /* r_tmp0, r_tmp1, r_tmp2 */ /* Sanity check: arena didn't overflow. */ - assert(ab->used <= ResolveLookAtArena_Words); + assert(ab.used <= ResolveLookAtArena_Words); } /* Emit the resolve_look_at bundle into the tape. Called once per frame from update(). @@ -277,6 +266,8 @@ I_ void resolve_look_at( , P3_S4* target , V3_S4* up_in ){ + // tb_emit_bundle(tb, slice_from_array(MipsAtom, smem.resolve_look_at_atom_addrs)); + /* Atom 0: input_and_sub — stages eye/up_in into scratchpad + computes fwd. */ tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); { tb_data(tb, u4_(target)); /* Binds_ResolveLookAtSub.target (C-side P3_S4*) */ @@ -285,18 +276,18 @@ I_ void resolve_look_at( tb_data(tb, u4_(smem.scratchpad)); /* Binds_ResolveLookAtScratch.scratch_base */ } - /* Atoms 1-5: NO tb_data — each chain atom uses r_scratch + hardcoded_offset internally (no tape-data pointers between atoms). + /* Atoms 1-5: Context carrier R_ResolveScratch (R_T4) is preserved across atoms. */ tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { } - tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { } - tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { } - tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { } - tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { } + // tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { } + // tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { } + // tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { } + // tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { } - /* Atom 6: populate_and_translate — only output pointer is the matrix destination. */ - tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); { - tb_data(tb, u4_(look_at)); /* Binds_ResolveLookAtPopAndTrans.look_at (MT3_S2S4*) */ - } + // /* Atom 6: populate_and_translate — only output pointer is the matrix destination. */ + // tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); { + // tb_data(tb, u4_(look_at)); /* Binds_ResolveLookAtPopAndTrans.look_at (MT3_S2S4*) */ + // } } FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); } @@ -351,46 +342,36 @@ void update(PrimitiveArena* pa, U4* ordering_buf) A2_S2 p; //??? S4 flag; //???? - // Camera Look at (Tape) + inline C11 fallback — bundle runs, then C11 inlines the look_at. - // Currently: bundle's atom 0 (input_and_sub) runs + C11 does the rest. As bundle atoms - // are incrementally fixed, the corresponding C11 lines get commented out. + if (0) { + camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0)); + } if (1) { tb.used = 0; tb_scope_run(& tb) { resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0)); } - - // RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE. - // Preconditions: eye != target, up_in not collinear with (target - eye). V3_S4 right, up, forward; V3_S4 ux, uy, uz; V3_S4 pos, off; - // forward = smem.cube.pos; sub_v3s4(& forward, smem.cam.pos); // RGA(Lengyel): Affine point - point = zero-weight direction. (now done by bundle atom 0) - // Read fwd from scratchpad[+0] (atom 0's output) - forward.x = u4_v(0x1F800000)[0]; - forward.y = u4_v(0x1F800000)[1]; - forward.z = u4_v(0x1F800000)[2]; - forward.pad = u4_v(0x1F800000)[3]; - // normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization. (now done by bundle atom 1) - // Read uz from scratchpad[+16] (atom 1's output) - uz.x = u4_v(0x1F800010)[0]; - uz.y = u4_v(0x1F800010)[1]; - uz.z = u4_v(0x1F800010)[2]; - uz.pad = u4_v(0x1F800010)[3]; + ResolveLookAtScratch_V scratch = C_scratch(ResolveLookAtScratch_V); - cross_v3s4(& uz, & v3s4(0, -fp_one, 0), & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis. - cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis. + // Atom 0: Works + forward = scratch->fwd; + + // Atom 1: + // normalize_v3s4(& forward, & uz); + uz = scratch->uz; + + cross_v3s4(& uz, & v3s4(0, -fp_one, 0), & right); normalize_v3s4(& right, & ux); + cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); - // RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows). smem.cam.look_at.m[0][0] = ux.x; smem.cam.look_at.m[0][1] = ux.y; smem.cam.look_at.m[0][2] = ux.z; smem.cam.look_at.m[1][0] = uy.x; smem.cam.look_at.m[1][1] = uy.y; smem.cam.look_at.m[1][2] = uy.z; smem.cam.look_at.m[2][0] = uz.x; smem.cam.look_at.m[2][1] = uz.y; smem.cam.look_at.m[2][2] = uz.z; pos = smem.cam.pos; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped). - // RGA(Lengyel): R * (-eye) is the full matrix translation column. - // Motor translator would store half this displacement in m.xyz; GTE consumes full column. mul_m3s2_v3s4(& smem.cam.look_at, & pos, & off); trans_m3s2( & smem.cam.look_at, & off); } @@ -498,7 +479,8 @@ GCC_OPTIMIZATION_DISABLE int main(void) { smem = (SMemory){0}; - smem.scratchpad = C_(U4_V, 0x1F800000); + // TODO(Ed): remove this field we don't need it in smem. + smem.scratchpad = C_(U4_V, Scratchpad_Loc); // smem.primitives.used = 0; // smem.active_buf_id = 0; smem.cam.pos = v3s4(500, -1000, -1500);