mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-09-08 09:19:06 +00:00
Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3faccfc283 | ||
|
|
1a0d417649 | ||
|
|
3301826f5c | ||
|
|
d9b9241e2c | ||
|
|
a16c727db2 |
BIN
Binary file not shown.
+1
-1
@@ -49,7 +49,7 @@ const DSL_KEYWORDS = new Set([
|
||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||
]);
|
||||
|
||||
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_"]);
|
||||
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
|
||||
|
||||
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
||||
|
||||
|
||||
@@ -56,7 +56,7 @@
|
||||
"name": "support.function.duffle.annotation"
|
||||
},
|
||||
"delay-slots": {
|
||||
"match": "\\b(LdSlot_|BdSlot_)\\b",
|
||||
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
|
||||
"name": "keyword.operator.duffle.delayslot"
|
||||
},
|
||||
"types": {
|
||||
|
||||
@@ -105,6 +105,16 @@ test("document-local declarations override an empty workspace index", () => {
|
||||
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
|
||||
});
|
||||
|
||||
test("delay slot markers share the tapeDelaySlot token", () => {
|
||||
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
|
||||
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
|
||||
|
||||
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
|
||||
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
|
||||
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
|
||||
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
|
||||
});
|
||||
|
||||
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
|
||||
const source = "atom_reads(R_A /* broken";
|
||||
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
||||
|
||||
+43
-6
@@ -58,6 +58,17 @@ WORD_COUNT(mac_yield_load, 1)
|
||||
, nop
|
||||
WORD_COUNT(mac_yield_tail, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_half_v3(tx, ty, tz, base, offset) \
|
||||
load_half(tx, base, offset + OA_(U2,[0])) \
|
||||
, load_half(ty, base, offset + OA_(U2,[1])) \
|
||||
, load_half(tz, base, offset + OA_(U2,[2]))
|
||||
WORD_COUNT(mac_load_half_v3, 3)
|
||||
|
||||
#define mac_load_v3s2(transfer, base, offset) \
|
||||
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_load_v3s2, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
|
||||
@@ -85,6 +96,17 @@ WORD_COUNT(mac_load_v3s4, 3)
|
||||
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_load_p3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_half_v3(tx, ty, tz, base, offset) \
|
||||
store_half(tx, base, offset + OA_(U2,[0])) \
|
||||
, store_half(ty, base, offset + OA_(U2,[1])) \
|
||||
, store_half(tz, base, offset + OA_(U2,[2]))
|
||||
WORD_COUNT(mac_store_half_v3, 3)
|
||||
|
||||
#define mac_store_v3s2(transfer, base, offset) \
|
||||
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||
WORD_COUNT(mac_store_v3s2, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_word_v3(tx, ty, tz, base, offset) \
|
||||
store_word(tx, base, offset + OA_(U4,[0])) \
|
||||
@@ -108,12 +130,27 @@ WORD_COUNT(mac_store_p3s4, 3)
|
||||
WORD_COUNT(mac_add_si_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
|
||||
sub_s(rds_x, rds_x, rt_x) \
|
||||
, sub_s(rds_y, rds_y, rt_y) \
|
||||
, sub_s(rds_z, rds_z, rt_z)
|
||||
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
|
||||
sub_s(dx, sx, tx) \
|
||||
, sub_s(dy, sy, ty) \
|
||||
, sub_s(dz, sz, tz)
|
||||
WORD_COUNT(mac_sub_s_v3, 3)
|
||||
|
||||
#define mac_sub_v3s4(d, s, t) \
|
||||
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
|
||||
WORD_COUNT(mac_sub_v3s4, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
|
||||
sub_s(ds_x, ds_x, tx) \
|
||||
, sub_s(ds_y, ds_y, ty) \
|
||||
, sub_s(ds_z, ds_z, tz)
|
||||
WORD_COUNT(mac_sub_s_v3_self, 3)
|
||||
|
||||
#define mac_sub_v3s4_self(ds, t) \
|
||||
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
|
||||
WORD_COUNT(mac_sub_v3s4_self, 3)
|
||||
|
||||
/* atom_dbg_skip */
|
||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||
@@ -250,8 +287,8 @@ WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
||||
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
||||
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
||||
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
||||
, DmaSlot_ nop_slot1 \
|
||||
, DmaSlot_ nop_slot2 \
|
||||
, GteDelay_ nop_slot1 \
|
||||
, GteDelay_ nop_slot2 \
|
||||
, gte_cmdw_gpf \
|
||||
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||
|
||||
@@ -25,7 +25,7 @@
|
||||
#pragma region duffle
|
||||
|
||||
|
||||
// --- atom: example_atom (10 words) ---
|
||||
// --- atom: example_atom_proc (10 words) ---
|
||||
|
||||
#define _atom_offset_example_atom_proc_skip 2
|
||||
|
||||
@@ -33,7 +33,7 @@ enum {
|
||||
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
|
||||
};
|
||||
|
||||
// --- atom: normalize_v3s4 (47 words) ---
|
||||
// --- atom: build_normalize_v3s4 (61 words) ---
|
||||
|
||||
#define _atom_offset_aligned_done_srav_path 3
|
||||
#define _atom_offset_srav_path_aligned_done 4
|
||||
|
||||
@@ -68,6 +68,7 @@ enum {
|
||||
|
||||
#define gp0_send(word) (HW_GP0[0] = (word))
|
||||
#define gp1_send(word) (HW_GP1[0] = (word))
|
||||
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
|
||||
|
||||
/* ============================================================================
|
||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||
|
||||
+17
-9
@@ -150,8 +150,8 @@ MipsAtomComp_Proc_(ab, {
|
||||
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||
gte_mv_to_data_r(to_ir2, C2_IR2),
|
||||
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
||||
DmaSlot_ nop_slot1,
|
||||
DmaSlot_ nop_slot2,
|
||||
GteDelay_ nop_slot1,
|
||||
GteDelay_ nop_slot2,
|
||||
gte_cmdw_gpf,
|
||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||
@@ -236,8 +236,13 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||
};
|
||||
|
||||
typedef Struct_(RegUse_normalize_v3s4_proc) {
|
||||
Reg const scratch; // Scratch base carrier.
|
||||
typedef Struct_(Binds_build_normalize_v3s4) {
|
||||
U4 scratch;
|
||||
U2 src_offset;
|
||||
U2 dst_offset;
|
||||
};
|
||||
typedef Struct_(RegUse_build_normalize_v3s4) {
|
||||
Reg scratch;
|
||||
Reg src_ptr;
|
||||
Reg dst_ptr;
|
||||
Reg recip_est; // |v|² sum + shift-input + sqrtbl[index]
|
||||
@@ -278,8 +283,11 @@ typedef Struct_(RegUse_normalize_v3s4_proc) {
|
||||
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
|
||||
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
|
||||
*/
|
||||
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_normalize_v3s4_proc r)
|
||||
internal MipsAtom* build_normalize_v3s4(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_build_normalize_v3s4 r)
|
||||
MipsAtom_Proc_(aa, {
|
||||
// load_word(r.scratch, R_TapePtr, O_(Binds_build_normalize_v3s4,scratch)),
|
||||
// add_ui_self(R_TapePtr, S_(Binds_build_normalize_v3s4)),
|
||||
|
||||
add_si(r.src_ptr, r.scratch, src_offset), /* r_src_ptr = &src */
|
||||
|
||||
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
|
||||
@@ -293,8 +301,8 @@ MipsAtom_Proc_(aa, {
|
||||
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
|
||||
add_u_self( r.norm, r.t3.mac1_scratch),
|
||||
add_u_self( r.norm, r.t4.mac2_scratch),
|
||||
gte_mv_to_data_r( r.norm, C2_LZCS), DmaSlot_ nop2,
|
||||
gte_mv_from_data_r(r.shift, C2_LZCR), DmaSlot_ nop,
|
||||
gte_mv_to_data_r( r.norm, C2_LZCS), GteDelay_ nop2,
|
||||
gte_mv_from_data_r(r.shift, C2_LZCR), GteDelay_ nop,
|
||||
|
||||
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
|
||||
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
|
||||
@@ -328,8 +336,8 @@ MipsAtom_Proc_(aa, {
|
||||
r.recip_est,
|
||||
r.t5.src_z, /* IR3 = src.z (reloaded) */
|
||||
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
|
||||
DmaSlot_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
|
||||
DmaSlot_ nop
|
||||
GteDelay_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
|
||||
GteDelay_ nop
|
||||
),
|
||||
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
|
||||
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
|
||||
|
||||
+1
-1
@@ -310,7 +310,7 @@ enum { _C2_TX_SUBS_ = 0
|
||||
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
||||
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
||||
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
||||
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
|
||||
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
|
||||
|
||||
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
||||
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
||||
|
||||
+38
-5
@@ -7,11 +7,22 @@
|
||||
|
||||
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||
|
||||
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
|
||||
|
||||
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
|
||||
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
|
||||
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
|
||||
|
||||
#pragma region MACs (Mips Atom Component)
|
||||
|
||||
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_half(tx, base, offset + OA_(U2,[0])),
|
||||
load_half(ty, base, offset + OA_(U2,[1])),
|
||||
load_half(tz, base, offset + OA_(U2,[2])),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
|
||||
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
|
||||
@@ -31,7 +42,15 @@ FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg
|
||||
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, U4 tx, U4 ty, U4 tz, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_half(tx, base, offset + OA_(U2,[0])),
|
||||
store_half(ty, base, offset + OA_(U2,[1])),
|
||||
store_half(tz, base, offset + OA_(U2,[2])),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||
|
||||
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_word(tx, base, offset + OA_(U4,[0])),
|
||||
store_word(ty, base, offset + OA_(U4,[1])),
|
||||
store_word(tz, base, offset + OA_(U4,[2])),
|
||||
@@ -47,12 +66,26 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
add_si(rt_z, base, O_(V3_S4,z)),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
sub_s(rds_x, rds_x, rt_x),
|
||||
sub_s(rds_y, rds_y, rt_y),
|
||||
sub_s(rds_z, rds_z, rt_z),
|
||||
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
|
||||
, Reg dx, Reg dy, Reg dz
|
||||
, Reg sx, Reg sy, Reg sz
|
||||
, Reg tx, Reg ty, Reg tz
|
||||
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
sub_s(dx, sx, tx),
|
||||
sub_s(dy, sy, ty),
|
||||
sub_s(dz, sz, tz),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
sub_s(ds_x, ds_x, tx),
|
||||
sub_s(ds_y, ds_y, ty),
|
||||
sub_s(ds_z, ds_z, tz),
|
||||
})
|
||||
|
||||
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
|
||||
|
||||
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
#pragma region hello_camera
|
||||
|
||||
|
||||
// --- atom: pad_input_cube_rotation (60 words) ---
|
||||
// --- atom: pad_input_cube_rotation (61 words) ---
|
||||
|
||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||
@@ -26,7 +26,7 @@ enum {
|
||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||
};
|
||||
|
||||
// --- atom: pad_input_cam (42 words) ---
|
||||
// --- atom: pad_input_cam (40 words) ---
|
||||
|
||||
#define _atom_offset_left_x_exit_left_x 3
|
||||
#define _atom_offset_right_x_exit_right_x 3
|
||||
@@ -44,7 +44,7 @@ enum {
|
||||
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
||||
};
|
||||
|
||||
// --- atom: cube_g4_face (76 words) ---
|
||||
// --- atom: cube_g4_face (75 words) ---
|
||||
|
||||
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||
|
||||
@@ -94,6 +94,26 @@ MipsAtomComp_Proc_(ab, {
|
||||
#pragma region Atom Procs
|
||||
// Modular Atoms
|
||||
|
||||
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
||||
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
||||
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
||||
|
||||
#pragma region resolve_look_at
|
||||
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
||||
|
||||
typedef AtomBundle_(resolve_look_at) { MipsAtom*
|
||||
input_and_sub,
|
||||
normalize_fwd_uz,
|
||||
cross_uz_up_into_right,
|
||||
normalize_right_ux,
|
||||
cross_uz_ux_to_up,
|
||||
normalize_up_uy,
|
||||
populate,
|
||||
set_gte_mt3s2s4,
|
||||
matrix_vector,
|
||||
trans_matrix;
|
||||
};
|
||||
|
||||
enum {
|
||||
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
|
||||
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
|
||||
@@ -119,16 +139,13 @@ typedef Struct_(ResolveLookAtScratch) {
|
||||
V3_S4 up_in; /* offset +128 (16 bytes) */
|
||||
};
|
||||
|
||||
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
||||
|
||||
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
|
||||
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||
ResolveLookAtScratch* scratchpad;
|
||||
};
|
||||
|
||||
typedef Struct_(RegUse_resolve_look_at__input_and_sub_proc) {
|
||||
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
|
||||
Reg scratch;
|
||||
Reg target; Reg eye; Reg up_in;
|
||||
Reg t0; Reg t1; Reg t2; Reg t3; Reg t4;
|
||||
@@ -151,26 +168,27 @@ typedef Struct_(RegUse_resolve_look_at__input_and_sub_proc) {
|
||||
* R_V0 : hardcoded (load eye.z / target.z)
|
||||
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
|
||||
*/
|
||||
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa, RegUse_resolve_look_at__input_and_sub_proc r)
|
||||
// internal MipsAtom* resolve_look_at_input_and_sub(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
|
||||
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
|
||||
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||
load_word(r.target, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||
load_word(r.eye, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||
load_word(r.up_in, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||
load_word(r.scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||
|
||||
/* Stage up_in.x/y/z into the scratchpad. */
|
||||
mac_load_word_v3( r.t0, r.t1, r.t2, r.up_in, 0),
|
||||
mac_load_word_v3( r.t0, r.t1, r.t2, r.up_in, 0), LdSlot_
|
||||
mac_store_word_v3(r.t0, r.t1, r.t2, r.scratch, O_(ResolveLookAtScratch,up_in)),
|
||||
|
||||
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
||||
mac_load_word_v3( r.t0, r.t1, r.t2, r.eye, 0),
|
||||
mac_load_word_v3( r.t0, r.t1, r.t2, r.eye, 0), LdSlot_
|
||||
mac_store_word_v3(r.t0, r.t1, r.t2, r.scratch, O_(ResolveLookAtScratch,eye)),
|
||||
|
||||
/* Compute fwd = target - eye. */
|
||||
// mac_load_p3s4(t3, R_AT, t4, r.eye, 0),
|
||||
mac_load_word_v3(r.t3, R_AT, r.t4, r.target, 0),
|
||||
mac_sub_v3s4(
|
||||
mac_load_word_v3(r.t3, R_AT, r.t4, r.target, 0), LdSlot_
|
||||
mac_sub_s_v3_self(
|
||||
r.t3, R_AT, r.t4,
|
||||
r.t0, r.t1, r.t2),
|
||||
mac_store_word_v3(r.t3, R_AT, r.t4, r.scratch, O_(ResolveLookAtScratch,fwd)),
|
||||
@@ -178,7 +196,7 @@ atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
typedef Struct_(RegUse_resolve_look_at__cross_uz_up_into_right_proc) {
|
||||
typedef Struct_(RegUse_resolve_look_at_cross_uz_up_into_right) {
|
||||
Reg scratch;
|
||||
Reg a; Reg b; Reg c; /* load a.x/y/z; result out.x/y/z */
|
||||
Reg d; /* load b.x */
|
||||
@@ -189,8 +207,9 @@ typedef Struct_(RegUse_resolve_look_at__cross_uz_up_into_right_proc) {
|
||||
Reg t0;
|
||||
};
|
||||
/* Atom 2: cross uz × up_in → right. */
|
||||
internal MipsAtom* resolve_look_at__cross_uz_up_into_right_proc(AtomArena_R aa,
|
||||
RegUse_resolve_look_at__cross_uz_up_into_right_proc r
|
||||
// internal MipsAtom* AtomBundleEntry_(resolve_look_at, cross_uz_up_to_right)(AtomArena_R aa,
|
||||
internal MipsAtom* resolve_look_at_cross_uz_up_into_right(AtomArena_R aa,
|
||||
RegUse_resolve_look_at_cross_uz_up_into_right r
|
||||
) MipsAtom_Proc_(aa, {
|
||||
/* FIX: build packed RT22+RT33 with proper sign extension. */
|
||||
add_si(r.g, r.scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||
@@ -199,7 +218,7 @@ internal MipsAtom* resolve_look_at__cross_uz_up_into_right_proc(AtomArena_R aa,
|
||||
nop,
|
||||
|
||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||
mac_load_word_v3(r.a, r.b, r.c, r.g, 0),
|
||||
mac_load_word_v3(r.a, r.b, r.c, r.g, 0), LdSlot_
|
||||
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
|
||||
mac_load_word_v3(r.d, R_AT, r.t0, r.h, 0), LdSlot_ // (taken by gte_mv_from_ctrl_r)
|
||||
|
||||
@@ -214,19 +233,19 @@ internal MipsAtom* resolve_look_at__cross_uz_up_into_right_proc(AtomArena_R aa,
|
||||
* The $2 and $4 writes don't clobber each other (separate registers).
|
||||
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
|
||||
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
|
||||
* This is libpsyx's OuterProduct12 convention EXACTLY. */
|
||||
* This is libpsyx's OuterProduct12 convention. */
|
||||
gte_mv_to_ctrl_r(r.b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
|
||||
gte_mv_to_ctrl_r(r.c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
|
||||
|
||||
/* Load uz into the RT diagonal. */
|
||||
gte_mv_to_ctrl_r(r.a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
|
||||
DmaSlot_ nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
GteDelay_ nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
|
||||
/* Load up_in into IR (the second operand for OP). */
|
||||
gte_mv_to_data_r(r.d, C2_IR1), /* IR1 = up_in.x */
|
||||
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
|
||||
gte_mv_to_data_r(r.t0, C2_IR3), /* IR3 = up_in.z */
|
||||
DmaSlot_ nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
GteDelay_ nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
|
||||
gte_cmdw_cross, /* OP: MAC1/2/3 = uz × up_in
|
||||
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
|
||||
@@ -243,7 +262,7 @@ internal MipsAtom* resolve_look_at__cross_uz_up_into_right_proc(AtomArena_R aa,
|
||||
|
||||
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
|
||||
mac_gte_mv_from_data_r_mac123(r.a, r.b, r.c),
|
||||
DmaSlot_ nop, /* MFC2 retirement */
|
||||
GteDelay_ nop, /* MFC2 retirement */
|
||||
|
||||
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
|
||||
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
|
||||
@@ -254,74 +273,61 @@ internal MipsAtom* resolve_look_at__cross_uz_up_into_right_proc(AtomArena_R aa,
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
typedef Struct_(RegUse_resolve_look_at__cross_uz_ux_to_up_proc) {
|
||||
Reg const scratch; /* pinned T4 */
|
||||
Reg_(V3_S4) a; /* uz components, then MAC / out */
|
||||
Reg_(V3_S4) b; /* ux components */
|
||||
union { Reg t0, up; }; /* &up, dedicated */
|
||||
union { Reg t1, uz, rt11; }; /* &uz, then RT11 save */
|
||||
union { Reg t2, ux, rt22; }; /* &ux, then RT22 save */
|
||||
};
|
||||
/* Atom 4: cross uz × ux → up. */
|
||||
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
|
||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||
, U4 r_d /* load b.x */
|
||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
|
||||
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa,
|
||||
RegUse_resolve_look_at__cross_uz_ux_to_up_proc r
|
||||
) MipsAtom_Proc_(aa, {
|
||||
/* Compute the three scratch pointers from r_scratch. */
|
||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
|
||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
|
||||
nop,
|
||||
add_si(r.uz, r.scratch, O_(ResolveLookAtScratch,uz)), /* r.uz = &uz */
|
||||
add_si(r.ux, r.scratch, O_(ResolveLookAtScratch,ux)), /* r.ux = &ux */
|
||||
add_si(r.up, r.scratch, O_(ResolveLookAtScratch,up)), /* r.up = &up (out) */
|
||||
|
||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||
load_word(r_a, r_g, O_(V3_S4,x)),
|
||||
load_word(r_b, r_g, O_(V3_S4,y)),
|
||||
load_word(r_c, r_g, O_(V3_S4,z)),
|
||||
nop,
|
||||
|
||||
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
|
||||
load_word(r_d, r_h, O_(V3_S4,x)),
|
||||
load_word(R_AT, r_h, O_(V3_S4,y)),
|
||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||
nop,
|
||||
mac_load_v3s4(r.a, r.uz, 0), LdSlot_
|
||||
mac_load_v3s4(r.b, r.ux, 0), LdSlot_ /* taken by gte_mv_from_ctrl_r */
|
||||
|
||||
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
|
||||
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
|
||||
* ctc2 RT restore. */
|
||||
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux, ctc2 RT restore. */
|
||||
|
||||
/* Save the two RT control-register slots OP will clobber (reusing
|
||||
* r_g/r_h — they're no longer needed as scratch pointers). */
|
||||
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
|
||||
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
|
||||
/* Save the two RT control-register slots OP will clobber (reusing r.uz/r.ux — they're no longer needed as scratch pointers). */
|
||||
gte_mv_from_ctrl_r(r.rt11, gte_cr_RT11), /* r.rt11 = C2 $0 (RT11|RT12) */
|
||||
gte_mv_from_ctrl_r(r.rt22, gte_cr_RT22), /* r.rt22 = C2 $4 (RT22|RT33) */
|
||||
|
||||
/* Load uz into the RT diagonal — same packing as atom 1.
|
||||
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
||||
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
|
||||
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
|
||||
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
|
||||
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
|
||||
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
|
||||
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
|
||||
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
|
||||
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4).
|
||||
* OP reads RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high. */
|
||||
gte_mv_to_ctrl_r(r.a.y, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
|
||||
gte_mv_to_ctrl_r(r.a.z, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
|
||||
gte_mv_to_ctrl_r(r.a.x, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
|
||||
GteDelay_ nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
|
||||
/* Load ux into the IR registers (the second operand for OP). */
|
||||
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
|
||||
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
|
||||
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
|
||||
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
gte_mv_to_data_r(r.b.x, C2_IR1), /* IR1 = ux.x */
|
||||
gte_mv_to_data_r(r.b.y, C2_IR2), /* IR2 = ux.y */
|
||||
gte_mv_to_data_r(r.b.z, C2_IR3), /* IR3 = ux.z */
|
||||
GteDelay_ nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||
|
||||
gte_cmdw_outer_product,
|
||||
gte_cmdw_cross,
|
||||
|
||||
/* Restore the RT slots we clobbered. */
|
||||
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
|
||||
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
|
||||
gte_mv_to_ctrl_r(r.rt11, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
|
||||
gte_mv_to_ctrl_r(r.rt22, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
|
||||
|
||||
mac_gte_mv_from_data_r_mac123(r.a.x, r.a.y, r.a.z),
|
||||
GteDelay_ nop,
|
||||
|
||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||
gte_mv_from_data_r(r_b, C2_MAC2),
|
||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||
nop,
|
||||
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
|
||||
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
|
||||
shift_aright(r_a, r_a, 12),
|
||||
shift_aright(r_b, r_b, 12),
|
||||
shift_aright(r_c, r_c, 12),
|
||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||
store_word(r_c, r_f, O_(V3_S4,z)),
|
||||
mac_shift_aright_v3_self(r.a.x, r.a.y, r.a.z, 12),
|
||||
mac_store_v3s4(r.a, r.up, 0),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
@@ -329,99 +335,47 @@ internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_
|
||||
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
|
||||
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
|
||||
};
|
||||
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye).
|
||||
*
|
||||
* GPR codes (assigned by resolve_look_at_init):
|
||||
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
||||
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
|
||||
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
|
||||
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
|
||||
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
|
||||
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
|
||||
*
|
||||
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
|
||||
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
|
||||
*
|
||||
* Struct layout (per duffle/math.h):
|
||||
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
||||
typedef Struct_(RegUse_resolve_look_at__populate_proc) {
|
||||
Reg const scratch;
|
||||
Reg look_at;
|
||||
Reg_(V3_S4) row; /* one matrix row, reused */
|
||||
Reg ux;
|
||||
Reg uy;
|
||||
Reg uz;
|
||||
};
|
||||
/* Atom 6a: write look_at->m[][] from ux/uy/uz as S2. Zero t[].
|
||||
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; }
|
||||
* m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
||||
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
|
||||
*
|
||||
* Translation column: GTE MVMVA with the world rotation matrix pre-set
|
||||
* (helper emits set_gte_world before the bundle, per the bundle design).
|
||||
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
|
||||
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
|
||||
*/
|
||||
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
|
||||
, U4 r_look_at
|
||||
, U4 r_scratch
|
||||
, U4 r_pux, U4 r_puy, U4 r_puz
|
||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa,
|
||||
RegUse_resolve_look_at__populate_proc r
|
||||
) MipsAtom_Proc_(aa, {
|
||||
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
|
||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||
|
||||
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
|
||||
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
|
||||
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
|
||||
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
|
||||
nop,
|
||||
add_si(r.ux, r.scratch, O_(ResolveLookAtScratch,ux)), /* r.ux = &ux */ LdSlot_
|
||||
add_si(r.uy, r.scratch, O_(ResolveLookAtScratch,uy)), /* r.uy = &uy */ LdSlot_
|
||||
add_si(r.uz, r.scratch, O_(ResolveLookAtScratch,uz)), /* r.uz = &uz */ LdSlot_
|
||||
|
||||
/* ── m[0] = (S2)ux ── */
|
||||
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
|
||||
nop,
|
||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
|
||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
|
||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
|
||||
|
||||
/* ── m[1] = (S2)uy ── */
|
||||
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
|
||||
nop,
|
||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
|
||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
|
||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
|
||||
|
||||
/* ── m[2] = (S2)uz ── */
|
||||
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
|
||||
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
|
||||
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
|
||||
nop,
|
||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
|
||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
|
||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
|
||||
mac_load_v3s4(r.row, r.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4,m[0])),
|
||||
mac_load_v3s4(r.row, r.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4,m[1])),
|
||||
mac_load_v3s4(r.row, r.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4,m[2])),
|
||||
|
||||
/* Zero t[0..2] — atom 6c writes the final values here. */
|
||||
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
|
||||
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
|
||||
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
|
||||
|
||||
mac_store_v3s4(v3s4_R_0(), r.look_at, O_(MT3_S2S4,t)),
|
||||
mac_yield()
|
||||
})
|
||||
|
||||
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
|
||||
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
|
||||
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
|
||||
* Stores off to scratch+96 (overwriting the packed pos).
|
||||
typedef Struct_(RegUse_resolve_look_at__matrix_vector_proc) {
|
||||
Reg const scratch;
|
||||
Reg look_at;
|
||||
Reg eye; /* &scratch.eye; store dest for off */
|
||||
Reg_(V3_S4) v; /* RT words, then -eye, then off */
|
||||
};
|
||||
/* Atom 6b: off = look_at.m * (-eye) >> 12. Stores off over scratch.eye.
|
||||
*
|
||||
* GPR codes (assigned by resolve_look_at_init):
|
||||
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
||||
* r_peye : pointer to eye (slot +96, reused as off destination)
|
||||
* r_tmp0/1/2: -eye + GTE transfer scratch
|
||||
*
|
||||
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
|
||||
*/
|
||||
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
|
||||
, U4 r_scratch
|
||||
, U4 r_peye
|
||||
, U4 r_look_at
|
||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||
) MipsAtom_Proc_(aa, {
|
||||
/* === EXACT C11 ApplyMatrixLV replication ===
|
||||
* The C11 does:
|
||||
* C11 ApplyMatrixLV:
|
||||
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
||||
* 2. lw v.x/y/z from memory
|
||||
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
|
||||
@@ -430,64 +384,32 @@ internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
|
||||
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
||||
* 7. mfc2 MACs
|
||||
* 8. Combine: (pass1 << 3) + pass2
|
||||
*
|
||||
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
|
||||
* The combine simplifies: result = 0 + pass2 = pass2.
|
||||
* So we skip the S15 decomposition and just do pass 2 directly.
|
||||
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
|
||||
*/
|
||||
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa,
|
||||
RegUse_resolve_look_at__matrix_vector_proc r
|
||||
) MipsAtom_Proc_(aa, {
|
||||
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||
|
||||
/* Pop look_at* from tape. */
|
||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||
/* Load RT from look_at.m into C2[0..4]. Packed S2 pairs, same as set_gte_mt3s2s4. */
|
||||
load_word( r.v.x, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_ add_si(r.eye, r.scratch, O_(ResolveLookAtScratch,eye)), /* r.eye = &eye */
|
||||
load_word( r.v.y, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.v.x, gte_cr_RT11),
|
||||
load_word( r.v.z, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.v.y, gte_cr_RT12),
|
||||
load_word( r.v.x, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.v.z, gte_cr_RT13),
|
||||
load_half_u(r.v.y, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.v.x, gte_cr_RT21),
|
||||
/* pos = -eye. The three loads also retire the last CTC2. */ gte_mv_to_ctrl_r(r.v.y, gte_cr_RT22),
|
||||
GteDelay_ mac_load_p3s4(r.v, r.eye, 0), LdSlot_ mac_sub_v3s4(r.v, v3s4_R_0(), r.v), /* pos.x = -eye.x */
|
||||
/* mtc2 pos (as S16) to IR1/2/3. The GTE takes low 16 bits. pos fits in S16. For negative pos, the 32-bit sign-extended value's low 16 bits = correct S16. */
|
||||
gte_mv_to_data_r(r.v.x, C2_IR1),
|
||||
gte_mv_to_data_r(r.v.y, C2_IR2),
|
||||
gte_mv_to_data_r(r.v.z, C2_IR3),
|
||||
GteDelay_ nop2,
|
||||
|
||||
/* r_peye = &eye (slot +96, reused as off destination). */
|
||||
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||
nop,
|
||||
|
||||
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
|
||||
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
|
||||
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
|
||||
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
|
||||
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
|
||||
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
|
||||
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
|
||||
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
|
||||
|
||||
/* Load pos = -eye after the matrix load releases r_tmp0. */
|
||||
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
|
||||
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
|
||||
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
|
||||
nop,
|
||||
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
|
||||
sub_u(r_tmp1, R_0, r_tmp1),
|
||||
sub_u(r_tmp2, R_0, r_tmp2),
|
||||
|
||||
/* === mtc2 pos (as S16) to IR1/2/3 ===
|
||||
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
|
||||
* 32-bit sign-extended value's low 16 bits = correct S16. */
|
||||
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
|
||||
* gives the correct S16 value (sign bit preserved). */
|
||||
/* r_tmp0/1/2 already have pos values. */
|
||||
gte_mv_to_data_r(r_tmp0, C2_IR1),
|
||||
gte_mv_to_data_r(r_tmp1, C2_IR2),
|
||||
gte_mv_to_data_r(r_tmp2, C2_IR3),
|
||||
nop2, /* MTC2 retirement (2 slots) */
|
||||
|
||||
/* === MVMVA pass 2 — C11 ApplyMatrixLV command ===
|
||||
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
||||
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
||||
gte_cmdw_mvmva_c11_pass2,
|
||||
nop, /* GTE interlock */
|
||||
|
||||
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
|
||||
gte_mv_from_data_r(r_tmp0, C2_MAC1),
|
||||
gte_mv_from_data_r(r_tmp1, C2_MAC2),
|
||||
gte_mv_from_data_r(r_tmp2, C2_MAC3),
|
||||
nop,
|
||||
|
||||
/* === Store off → scratch+96 (overwriting pos) === */
|
||||
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
|
||||
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
|
||||
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
|
||||
gte_cmdw_mvmva_c11_pass2, GteDelay_ nop,
|
||||
mac_gte_mv_from_data_r_mac123(r.v.x, r.v.y, r.v.z), GteDelay_ nop,
|
||||
mac_store_v3s4(r.v, r.eye, 0),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
@@ -507,19 +429,15 @@ I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa
|
||||
, U4 r_look_at, U4 r_scratch, U4 r_off_ptr
|
||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||
) MipsAtom_Proc_(aa, {
|
||||
/* Pop look_at* from tape. */
|
||||
// load_word(r_Vlook_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||
// add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||
|
||||
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */
|
||||
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||
nop,
|
||||
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)), nop,
|
||||
|
||||
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */
|
||||
mac_trans_mt3s3s4(r_look_at, r_off_ptr, r_tmp0, r_tmp1, r_tmp2),
|
||||
|
||||
mac_yield()
|
||||
})
|
||||
#pragma endregion resolve_look_at
|
||||
|
||||
#pragma endregion Atom Procs
|
||||
|
||||
@@ -651,15 +569,15 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
||||
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0. */
|
||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
|
||||
// Note(Ed): Potential op with delay slot?
|
||||
|
||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, 30),
|
||||
add_si( R_T3, R_T3, 5),
|
||||
@@ -668,8 +586,8 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
||||
atom_label(exit_dpad_left)
|
||||
|
||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
|
||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||
add_si( R_T4, R_T4, -30),
|
||||
add_si( R_T3, R_T3, -5),
|
||||
@@ -679,7 +597,7 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
||||
|
||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
|
||||
|
||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||
@@ -688,14 +606,14 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
||||
|
||||
atom_label(dead_check_upper)
|
||||
/* left_x >= 0x70 → check upper bound. */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
|
||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
|
||||
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
||||
|
||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
|
||||
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||
mac_yield_load(),
|
||||
BdSlot_ mac_yield_load(), LdSlot_
|
||||
|
||||
atom_label(dead_low_active)
|
||||
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
@@ -706,18 +624,18 @@ atom_label(dead_low_active)
|
||||
|
||||
/* R_T4 = cube_delta */
|
||||
shift_aright(R_T4, R_T3, 2),
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
jump_rel(atom_offset(end_low, exit_stick)),
|
||||
mac_yield_load(),
|
||||
BdSlot_ mac_yield_load(), LdSlot_
|
||||
|
||||
atom_label(dead_high_active)
|
||||
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||
@@ -727,18 +645,18 @@ atom_label(dead_high_active)
|
||||
/* delta = 0x80 - left_x (signed negative). */
|
||||
|
||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||
|
||||
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||
shift_aright(R_T4, R_T3, 5),
|
||||
add_u( R_T0, R_T0, R_T4),
|
||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||
|
||||
atom_label(no_jump_fallthrough)
|
||||
mac_yield_load(),
|
||||
mac_yield_load(), LdSlot_
|
||||
|
||||
atom_label(exit_stick)
|
||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||
@@ -760,14 +678,14 @@ internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
||||
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
||||
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||
|
||||
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
||||
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
|
||||
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
|
||||
load_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
|
||||
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
||||
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ mac_yield_load(),
|
||||
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ mac_yield_load(), LdSlot_
|
||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||
atom_label(exit_left_x)
|
||||
|
||||
@@ -826,7 +744,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
@@ -839,20 +757,20 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||
|
||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
||||
GteDelay_ nop, gte_cmdw_rotate_translate_perspective_triple,
|
||||
gte_cmdw_nclip,
|
||||
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ nop,
|
||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||
* harmless because the OT entry that points to this prim is created later. */
|
||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
|
||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||
|
||||
mac_gte_store_g4_p012(R_PrimCursor),
|
||||
@@ -864,7 +782,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||
set_lt_u( R_AT, R_T1, R_AT),
|
||||
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
|
||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||
mac_format_g4_color(R_PrimCursor,
|
||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||
@@ -896,7 +814,7 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
|
||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||
mac_yield()
|
||||
};
|
||||
|
||||
@@ -943,7 +861,7 @@ internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitive
|
||||
, atom_writes(R_TapePtr)
|
||||
){
|
||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
|
||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||
/* Calculate byte offset and store directly back to RAM */
|
||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||
|
||||
@@ -60,8 +60,12 @@ enum {
|
||||
enum {
|
||||
Scratchpad_Len = 1024,
|
||||
MemTape_Len = 512,
|
||||
|
||||
ResolveLookAtArena_Words = 1024,
|
||||
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
||||
|
||||
CT_InitAtomMem_Words = Kilo_(4),
|
||||
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
|
||||
};
|
||||
typedef Struct_(SMemory) {
|
||||
PrimitiveArena primitives;
|
||||
@@ -85,8 +89,14 @@ typedef Struct_(SMemory) {
|
||||
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
||||
U4_V scratchpad; // d-cache
|
||||
|
||||
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
|
||||
MipsAtom* normalize_v3s4;
|
||||
// TODO(Ed): Convert normalize_v3s4 to a generic atom?
|
||||
// This would allow us to reduce specializations with the loss being some cycles to loading registers.
|
||||
// The cost would be 3 loads (scratch, src_ptr, dst_offset) from tape and
|
||||
|
||||
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
||||
MipsAtom* resolve_look_at_atom_addrs[10];
|
||||
MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
|
||||
};
|
||||
global SMemory smem;
|
||||
extern SMemory smem;
|
||||
@@ -131,28 +141,20 @@ I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4*
|
||||
}
|
||||
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||
|
||||
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
|
||||
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
|
||||
* share the GENERIC normalize_v3s4_proc from gte.atom.c
|
||||
* 0: resolve_look_at__input_and_sub_proc
|
||||
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
|
||||
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
|
||||
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
|
||||
* 4: resolve_look_at__cross_uz_ux_to_up_proc
|
||||
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
|
||||
* 6: resolve_look_at__populate_and_translate_proc
|
||||
*/
|
||||
internal void resolve_look_at_init(void) {
|
||||
|
||||
|
||||
internal void compile_resolve_look_at(void) {
|
||||
/* Wrap the static arena in a MipsAtomBuilder. */
|
||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_atom_addrs));
|
||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_bundle));
|
||||
|
||||
U4 pin_mask = regfile_abi_mask | (1 << R_ResolveScratch);
|
||||
RegFile rf = regfile(pin_mask);
|
||||
#define ralloc() regfile_alloc(& rf)
|
||||
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||
|
||||
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
|
||||
RegUse_(resolve_look_at__input_and_sub_proc) {
|
||||
tb_emit_(AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
|
||||
RegUse_(resolve_look_at_input_and_sub) {
|
||||
.scratch = R_ResolveScratch,
|
||||
.target = ralloc(),
|
||||
.eye = ralloc(),
|
||||
@@ -163,14 +165,14 @@ internal void resolve_look_at_init(void) {
|
||||
.t3 = ralloc(),
|
||||
.t4 = ralloc(),
|
||||
}
|
||||
);
|
||||
));
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 1: normalize fwd→uz === */
|
||||
U2 src_offset = O_(ResolveLookAtScratch, fwd);
|
||||
U2 dst_offset = O_(ResolveLookAtScratch, uz);
|
||||
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
|
||||
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
|
||||
smem.resolve_look_at_bundle[1] = build_normalize_v3s4(& ab,
|
||||
src_offset, dst_offset, RegUse_(build_normalize_v3s4){
|
||||
.scratch = R_ResolveScratch,
|
||||
.src_ptr = ralloc(),
|
||||
.dst_ptr = ralloc(),
|
||||
@@ -185,8 +187,9 @@ internal void resolve_look_at_init(void) {
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 2: cross uz×up_in→right === */
|
||||
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_into_right_proc(& ab,
|
||||
RegUse_(resolve_look_at__cross_uz_up_into_right_proc) {
|
||||
// smem.resolve_look_at_bundle[2] = AtomBundleEntry_(resolve_look_at,cross_uz_up_to_right)(& ab,
|
||||
smem.resolve_look_at_bundle[2] = resolve_look_at_cross_uz_up_into_right(& ab,
|
||||
RegUse_(resolve_look_at_cross_uz_up_into_right) {
|
||||
.scratch = R_ResolveScratch,
|
||||
.a = ralloc(),
|
||||
.b = ralloc(),
|
||||
@@ -197,95 +200,91 @@ internal void resolve_look_at_init(void) {
|
||||
.t2 = ralloc(),
|
||||
.t0 = ralloc(),
|
||||
});
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 3: normalize right→ux === */
|
||||
src_offset = O_(ResolveLookAtScratch, right);
|
||||
dst_offset = O_(ResolveLookAtScratch, ux);
|
||||
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
|
||||
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
|
||||
smem.resolve_look_at_bundle[3] = build_normalize_v3s4(& ab,
|
||||
src_offset, dst_offset, RegUse_(build_normalize_v3s4){
|
||||
.scratch = R_ResolveScratch,
|
||||
.src_ptr = R_T0,
|
||||
.dst_ptr = R_T1,
|
||||
.recip_est = R_T6,
|
||||
.norm = R_T7,
|
||||
.shift = R_V0,
|
||||
.src_x = R_T2,
|
||||
.t3 = R_T3,
|
||||
.t4 = R_T5,
|
||||
.t5 = R_V1,
|
||||
.src_ptr = ralloc(),
|
||||
.dst_ptr = ralloc(),
|
||||
.recip_est = ralloc(),
|
||||
.norm = ralloc(),
|
||||
.shift = ralloc(),
|
||||
.src_x = ralloc(),
|
||||
.t3 = ralloc(),
|
||||
.t4 = ralloc(),
|
||||
.t5 = ralloc(),
|
||||
});
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 4: cross uz×ux→up === */
|
||||
U4 r_a_4 = R_T0;
|
||||
U4 r_b_4 = R_T1;
|
||||
U4 r_c_4 = R_T2;
|
||||
U4 r_d_4 = R_T3;
|
||||
U4 r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
|
||||
U4 r_g_4 = R_T6; /* a ptr = scratch+16 */
|
||||
U4 r_h_4 = R_T7; /* b ptr = scratch+48 */
|
||||
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
|
||||
R_ResolveScratch,
|
||||
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
|
||||
smem.resolve_look_at_bundle[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
|
||||
RegUse_(resolve_look_at__cross_uz_ux_to_up_proc){
|
||||
.scratch = R_ResolveScratch,
|
||||
.a = ralloc_v3(), /* T0 T1 T2 */
|
||||
.b = ralloc_v3(), /* T3 T5 T6 */
|
||||
.t0 = ralloc(), /* T7 = up */
|
||||
.t1 = ralloc(), /* V0 = uz / rt11 */
|
||||
.t2 = ralloc(), /* V1 = ux / rt22 */
|
||||
});
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 5: normalize up→uy === */
|
||||
src_offset = O_(ResolveLookAtScratch, up);
|
||||
dst_offset = O_(ResolveLookAtScratch, uy);
|
||||
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
|
||||
smem.resolve_look_at_bundle[5] = build_normalize_v3s4(& ab,
|
||||
src_offset, dst_offset,
|
||||
RegUse_(normalize_v3s4_proc){
|
||||
RegUse_(build_normalize_v3s4){
|
||||
.scratch = R_ResolveScratch,
|
||||
.src_ptr = R_T0,
|
||||
.dst_ptr = R_T1,
|
||||
.recip_est = R_T6,
|
||||
.norm = R_T7,
|
||||
.shift = R_V0,
|
||||
.src_x = R_T2,
|
||||
.t3 = R_T3,
|
||||
.t4 = R_T5,
|
||||
.t5 = R_V1,
|
||||
.src_ptr = ralloc(),
|
||||
.dst_ptr = ralloc(),
|
||||
.recip_est = ralloc(),
|
||||
.norm = ralloc(),
|
||||
.shift = ralloc(),
|
||||
.src_x = ralloc(),
|
||||
.t3 = ralloc(),
|
||||
.t4 = ralloc(),
|
||||
.t5 = ralloc(),
|
||||
});
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
|
||||
U4 r_look_at_6a = R_T0; /* tape pop → look_at* */
|
||||
U4 r_scratch_6a = R_ResolveScratch;
|
||||
U4 r_pux_6a = R_T1;
|
||||
U4 r_puy_6a = R_T3;
|
||||
U4 r_puz_6a = R_T5;
|
||||
U4 r_tmp0_6a = R_T2;
|
||||
U4 r_tmp1_6a = R_T6;
|
||||
U4 r_tmp2_6a = R_V0;
|
||||
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
|
||||
r_look_at_6a, r_scratch_6a,
|
||||
r_pux_6a, r_puy_6a, r_puz_6a,
|
||||
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
|
||||
smem.resolve_look_at_bundle[6] = resolve_look_at__populate_proc(& ab,
|
||||
RegUse_(resolve_look_at__populate_proc){
|
||||
.scratch = R_ResolveScratch,
|
||||
.look_at = ralloc(), /* T0 */
|
||||
.row = ralloc_v3(), /* T1 T2 T3 */
|
||||
.ux = ralloc(), /* T5 = ux */
|
||||
.uy = ralloc(), /* T6 = uy */
|
||||
.uz = ralloc(), /* T7 = uz */
|
||||
});
|
||||
regfile_reset_to_mask(& rf, pin_mask);
|
||||
|
||||
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
|
||||
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
|
||||
* the matrix pointer (popped from tape). It does NOT need GPR
|
||||
* assignment from us — it has its own internal GPR usage.
|
||||
* We just take its address. */
|
||||
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
|
||||
smem.resolve_look_at_bundle[7] = (MipsAtom*) & set_gte_mt3s2s4;
|
||||
|
||||
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
|
||||
* Uses mac_apply_matrix_lv component macro which internally uses
|
||||
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
|
||||
* for the mfc2/store. We pass our GPRs. */
|
||||
U4 r_scratch_6b = R_ResolveScratch;
|
||||
U4 r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
|
||||
U4 r_look_at_6b = R_T0; /* tape pop → look_at* */
|
||||
U4 r_tmp0_6b = R_T2;
|
||||
U4 r_tmp1_6b = R_T3;
|
||||
U4 r_tmp2_6b = R_T5;
|
||||
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
|
||||
r_scratch_6b, r_peye_6b, r_look_at_6b,
|
||||
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
|
||||
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) === */
|
||||
smem.resolve_look_at_bundle[8] = resolve_look_at__matrix_vector_proc(& ab,
|
||||
RegUse_(resolve_look_at__matrix_vector_proc){
|
||||
.scratch = R_ResolveScratch,
|
||||
.look_at = ralloc(), /* T0 */
|
||||
.eye = ralloc(), /* T1 */
|
||||
.v = ralloc_v3(), /* T2 T3 T5 */
|
||||
});
|
||||
|
||||
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
|
||||
U4 r_look_at_6c = R_T0; /* tape pop → look_at* */
|
||||
U4 r_scratch_6c = R_ResolveScratch;
|
||||
U4 r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
|
||||
U4 r_tmp0_6c = R_T2;
|
||||
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
|
||||
smem.resolve_look_at_bundle[9] = resolve_look_at__trans_matrix_proc(& ab,
|
||||
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c, R_T3, R_T4);
|
||||
|
||||
/* Sanity check: arena didn't overflow. */
|
||||
@@ -304,36 +303,41 @@ internal void resolve_look_at_init(void) {
|
||||
* ----
|
||||
* 5 tb_data words total per frame.
|
||||
*/
|
||||
I_ void resolve_look_at(
|
||||
TapeBuilder_R tb
|
||||
I_ void resolve_look_at(TapeBuilder_R tb
|
||||
, MT3_S2S4* look_at
|
||||
, P3_S4* eye
|
||||
, P3_S4* target
|
||||
, V3_S4* up_in
|
||||
){
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[0]); {
|
||||
tb_data(tb, u4_(target));
|
||||
tb_data(tb, u4_(eye));
|
||||
tb_data(tb, u4_(up_in));
|
||||
tb_data(tb, u4_(smem.scratchpad));
|
||||
}
|
||||
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[1]); {
|
||||
// tb_data(tb, u4_(Scratchpad_Loc));
|
||||
}
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[2]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[3]); {
|
||||
// tb_data(tb, u4_(Scratchpad_Loc));
|
||||
}
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[4]); { }
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[5]); {
|
||||
// tb_data(tb, u4_(Scratchpad_Loc));
|
||||
}
|
||||
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[6]); {
|
||||
tb_data(tb, u4_(look_at));
|
||||
}
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[7]); {
|
||||
tb_data(tb, u4_(look_at));
|
||||
}
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[8]); {
|
||||
tb_data(tb, u4_(look_at));
|
||||
}
|
||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
|
||||
tb_emit(tb, smem.resolve_look_at_bundle[9]); {
|
||||
// tb_data(tb, u4_(look_at));
|
||||
}
|
||||
}
|
||||
@@ -527,8 +531,7 @@ int main(void)
|
||||
/* Direct BIOS: poll both ports during VBlank. */
|
||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||
|
||||
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
|
||||
resolve_look_at_init();
|
||||
compile_resolve_look_at();
|
||||
|
||||
/* Pinned registers for the GPU init atom. */
|
||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||
@@ -549,4 +552,3 @@ int main(void)
|
||||
return 0;
|
||||
}
|
||||
GCC_OPTIMIZATION_ENABLE
|
||||
|
||||
|
||||
+43
-18
@@ -1014,14 +1014,20 @@ end
|
||||
-- Section 7: domain tables
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
-- The annotation DSL has been reduced to a single annotation macro: atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...))
|
||||
-- All phase / region / cadence / async / resource / group tokens have been dropped.
|
||||
-- They may be reintroduced later as optional sub-calls of atom_info;
|
||||
-- For now, the parser only recognizes atom_info + its three sub-calls (atom_bind, atom_reads, atom_writes).
|
||||
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
|
||||
M.TAPE_ATOM_MACROS = {
|
||||
["atom_info"] = { kind = "info", binds = false },
|
||||
}
|
||||
|
||||
-- Empty C macros that prefix the next encoder. Zero words.
|
||||
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
|
||||
M.DELAY_MARKERS = {
|
||||
["GteDelay_"] = true,
|
||||
["LdSlot_"] = true,
|
||||
["BdSlot_"] = true,
|
||||
["DmaSlot_"] = true,
|
||||
}
|
||||
|
||||
-- GTE command-alias resolution table.
|
||||
--
|
||||
-- Maps every source-side GTE command macro to its canonical short ident.
|
||||
@@ -1328,6 +1334,12 @@ M.GTE_CR_ALIAS_GROUPS = {
|
||||
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
|
||||
}
|
||||
|
||||
-- Packed RT slots named by the gte.h packed-slot comment.
|
||||
-- first must be written before second.
|
||||
M.GTE_PACKED_SLOT_RELATIONS = {
|
||||
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
|
||||
}
|
||||
|
||||
-- Operand-class table for the COP2->GPR load-delay check.
|
||||
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
||||
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
||||
@@ -2455,6 +2467,15 @@ local function _project_emission_inner(root_body_entry, ctx_table)
|
||||
pos = (next_pos > pos) and next_pos or (pos + 1)
|
||||
goto continue_loop
|
||||
end
|
||||
if M.DELAY_MARKERS[ident] then
|
||||
local arg_pos = nil
|
||||
if consuming_encoder and consuming_paren then
|
||||
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
|
||||
end
|
||||
emit_marker("delay", ident, nil, tok_line, nil, nil, consuming_encoder, arg_pos)
|
||||
pos = after
|
||||
goto continue_loop
|
||||
end
|
||||
if ident ~= "atom_label" and ident ~= "atom_offset" then
|
||||
-- Ordinary ident; nothing to emit, step past the ident only.
|
||||
pos = after
|
||||
@@ -2601,9 +2622,18 @@ local function _project_emission_inner(root_body_entry, ctx_table)
|
||||
local function process_token(bt)
|
||||
local tok = M.trim(bt.tok or "")
|
||||
if tok == "" then return end
|
||||
local ident = M.read_ident(tok, 1) or "?"
|
||||
local ident, after = M.read_ident(tok, 1)
|
||||
if not ident then ident = "?" end
|
||||
local _, args = token_ident_and_args(tok)
|
||||
local tok_line = line_of(body_off + bt.rel) or 0
|
||||
if M.DELAY_MARKERS[ident] then
|
||||
emit_marker("delay", ident, nil, tok_line)
|
||||
local rest = M.trim(tok:sub(after or (#tok + 1)))
|
||||
if rest ~= "" then
|
||||
process_token({ tok = rest, rel = bt.rel })
|
||||
end
|
||||
return
|
||||
end
|
||||
-- embedded markers live only in non-marker tokens.
|
||||
-- Pass `ident` as the consuming instruction so `emit_embedded_markers` can compute each marker's arg position + record the consuming_encoder for the offsets pass.
|
||||
-- Canonicalize `jump_rel` to `branch_equal` (its preprocessor-expanded form) so the `consuming_encoder` metadata in marker records is canonical.
|
||||
@@ -2751,6 +2781,7 @@ end
|
||||
--- `nop2` is normalized to encoder `nop` (per the spec).
|
||||
--- * `atom_label(F)` markers: one `label` item with `name = "F"`, `word_index = current word_idx`; zero-width (does NOT advance word_idx).
|
||||
--- * `atom_offset(B, T)` markers: one `offset` item with `name = "B"`, `target = "T"`, `word_index = current word_idx`; zero-width.
|
||||
--- * Delay markers (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`): one `delay` item; zero-width. The following encoder is the next token.
|
||||
--- * `mac_X(...)` calls: emit `invoke_begin` (zero-width), recurse into the component body, emit `invoke_end` (zero-width).
|
||||
--- The component body's words land between the begin/end pair; one invocation record is allocated per call (monotonic ID per atom).
|
||||
--- * Unknown uncounted macros emit 1 opaque word + one warning per occurrence.
|
||||
@@ -2892,18 +2923,17 @@ end
|
||||
-------------------------------------------------------------------------------
|
||||
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
|
||||
--
|
||||
-- After the `sym` arg was dropped from MipsAtom_Proc_, the atom name is
|
||||
-- derived from the preceding `MipsAtom* X_proc(args)` function declaration.
|
||||
-- The atom name is the preceding `MipsAtom* ident(args)` function ident.
|
||||
-- This function walks backward from `before_pos` to find it.
|
||||
--
|
||||
-- Returns (raw_name, args_inner) or (nil, nil).
|
||||
-- raw_name — e.g. "normalize_v3s4" (the _proc suffix is stripped)
|
||||
-- Returns (raw_name, args_inner, func_ident, after_paren) or (nil, nil).
|
||||
-- raw_name — the function ident as written
|
||||
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
|
||||
-- after_paren — source position after the function `)`
|
||||
--
|
||||
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips
|
||||
-- whitespace + qualifiers (internal, I_, FI_, comments) until it finds an
|
||||
-- ident followed by "(". That ident is the function name (with _proc suffix);
|
||||
-- the suffix is stripped to get raw_name. The parens contents are the args.
|
||||
-- ident followed by "(". That ident is the name. The parens contents are the args.
|
||||
-------------------------------------------------------------------------------
|
||||
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
|
||||
local search_pos = 1
|
||||
@@ -2948,14 +2978,9 @@ function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
|
||||
-- check if the next non-ws char after ident is "("
|
||||
local next_pos = M.skip_ws_and_cmt(source, ident_end)
|
||||
if source:sub(next_pos, next_pos) == "(" then
|
||||
local inner = M.read_parens(source, next_pos)
|
||||
local inner, after_paren = M.read_parens(source, next_pos)
|
||||
if inner then
|
||||
local proc_suffix = "_proc"
|
||||
local atom_name = ident
|
||||
if #ident > #proc_suffix and ident:sub(-#proc_suffix) == proc_suffix then
|
||||
atom_name = ident:sub(1, #ident - #proc_suffix)
|
||||
end
|
||||
return atom_name, inner, ident
|
||||
return ident, inner, ident, after_paren
|
||||
end
|
||||
end
|
||||
-- ident not followed by "(" — it's a qualifier; skip it
|
||||
|
||||
@@ -612,13 +612,13 @@ function M.read_elf_sections(elf_path, section_names)
|
||||
end
|
||||
|
||||
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
|
||||
--- Returns a map `{name -> {addr, size_bytes}}` for every `code_<name>` symbol.
|
||||
--- Returns a map `{name -> {addr, size_bytes}}` for every defined symbol.
|
||||
---
|
||||
--- **Conventions:**
|
||||
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
|
||||
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
|
||||
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
||||
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
||||
--- - Keys are the ELF symbol names as written (the C ident).
|
||||
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
||||
---
|
||||
--- @param elf_path Path
|
||||
|
||||
@@ -16,7 +16,7 @@ define tape_atoms
|
||||
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||
end
|
||||
document tape_atoms
|
||||
List every tape atom symbol in the loaded ELF (code_<name>) with its .rodata address and word count.
|
||||
List every tape atom symbol in the loaded ELF with its .rodata address and word count.
|
||||
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
||||
end
|
||||
|
||||
|
||||
@@ -477,8 +477,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
||||
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
||||
local atoms = {}
|
||||
for _, a in ipairs(scan.atoms) do
|
||||
if a.kind == "atom" then
|
||||
atoms[#atoms + 1] = { line = a.line, name = a.raw_name }
|
||||
if a.kind == "atom" or a.kind == "atom_proc" then
|
||||
atoms[#atoms + 1] = { line = a.line, name = a.raw_name or a.name }
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
@@ -261,12 +261,12 @@ local function append_gdb_commands(lines, matched)
|
||||
for _, a in ipairs(matched) do
|
||||
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
||||
-- Use the per-atom convenience vars set above as printf args.
|
||||
lines[#lines + 1] = string.format(' printf " code_%%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||
lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||
a.idx, a.idx, a.idx)
|
||||
end
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = "document tape_atoms"
|
||||
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF (code_<name>) with .rodata addr + word count."
|
||||
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
|
||||
@@ -285,10 +285,10 @@ local function append_gdb_commands(lines, matched)
|
||||
for _, a in ipairs(matched) do
|
||||
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
||||
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
||||
lines[#lines + 1] = string.format(' printf " Breakpoint set at code_%s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||
lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
||||
lines[#lines + 1] = string.format(" Set a breakpoint at code_%s.", a.name)
|
||||
lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
|
||||
lines[#lines + 1] = "end"
|
||||
lines[#lines + 1] = ""
|
||||
end
|
||||
@@ -323,7 +323,7 @@ local function append_gdb_commands(lines, matched)
|
||||
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "atom: code_%%s\\n", $__atom_name_%d', a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
|
||||
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
||||
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
||||
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
||||
@@ -541,7 +541,7 @@ function M.render_atom_provenance(atom, wc, rel_path)
|
||||
return table.concat(lines, "\n") .. "\n"
|
||||
end
|
||||
|
||||
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
|
||||
--- Pass entry. For each source that declares at least one tape atom,
|
||||
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
||||
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||
---
|
||||
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
|
||||
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
|
||||
--- the ELF symbol is the C ident. Raw `MipsCode code_*` is leftover, not the atom rule.
|
||||
---
|
||||
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
||||
@@ -187,7 +187,7 @@ local function project_components(source, scan)
|
||||
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
|
||||
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
|
||||
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
|
||||
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
|
||||
-- `tb_emit` of the C ident, NOT inlined as a macro. Including `atom_proc` here
|
||||
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
|
||||
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
|
||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||
@@ -393,6 +393,10 @@ end
|
||||
--- @param cache table<string, integer>
|
||||
--- @return integer
|
||||
local function gp0_contrib_rec(name, comp_by_name, cache)
|
||||
if name:match("^insert_ot_tag") then
|
||||
cache[name] = 0
|
||||
return 0
|
||||
end
|
||||
if cache[name] ~= nil then return cache[name] end
|
||||
cache[name] = -1
|
||||
local cc = comp_by_name[name]
|
||||
@@ -408,8 +412,15 @@ local function gp0_contrib_rec(name, comp_by_name, cache)
|
||||
-- Nested `mac_X(...)` call: recurse.
|
||||
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
||||
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
||||
elseif ident == "gte_sw" then
|
||||
n = n + 1
|
||||
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
|
||||
if trimmed:find("R_PrimCursor", 1, true) then
|
||||
if trimmed:find("R_PrimCursor", 1, true)
|
||||
or trimmed:find("O_(Poly_", 1, true)
|
||||
or trimmed:find("r_prim_cursor", 1, true)
|
||||
or trimmed:find("r_primitive_cursor", 1, true)
|
||||
or trimmed:find("r_base", 1, true)
|
||||
then
|
||||
n = n + 1
|
||||
end
|
||||
end
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
---
|
||||
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
|
||||
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
|
||||
--- APPENDS synthetic DWARF line-program sequences for every `code_<name>` atom, EXTENDS the `.debug_aranges`
|
||||
--- APPENDS synthetic DWARF line-program sequences for every tape atom, EXTENDS the `.debug_aranges`
|
||||
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
|
||||
--- existing main compilation unit in `.debug_info` (no second compilation unit).
|
||||
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
|
||||
@@ -1344,7 +1344,7 @@ local function build_new_abbrev()
|
||||
attr( DW_AT_name, DW_FORM_string)
|
||||
.. attr(DW_AT_low_pc, DW_FORM_addr)
|
||||
.. attr(DW_AT_high_pc, DW_FORM_addr)
|
||||
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; lets gdb's symbol-table lookup resolve to our subprogram (not the gcc global `code_<name>` const U4 array)
|
||||
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; gdb resolves the subprogram, not the gcc global array
|
||||
|
||||
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
|
||||
attr( DW_AT_name, DW_FORM_string)
|
||||
@@ -1857,7 +1857,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
||||
end
|
||||
|
||||
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
|
||||
-- Subprogram names match nm symbols without a `code_` prefix.
|
||||
-- Subprogram names match the written C ident (the ELF symbol).
|
||||
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
|
||||
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
|
||||
for _, atom in ipairs(atom_table) do
|
||||
@@ -2313,5 +2313,6 @@ end
|
||||
M.compute_loclists_offsets_for_test = compute_loclists_offsets
|
||||
M.build_debug_loclists_section_for_test = build_debug_loclists_section
|
||||
M.tape_piece_size_for_test = tape_piece_size
|
||||
M.build_atom_table_for_test = build_atom_table
|
||||
|
||||
return M
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
--- passes/offsets.lua — Branch-offset generator.
|
||||
---
|
||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
||||
--- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
|
||||
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
|
||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||
---
|
||||
|
||||
+215
-20
@@ -259,6 +259,27 @@ local function slot_suffix(key)
|
||||
return key:match("([^:]+)$")
|
||||
end
|
||||
|
||||
local function decl_names(view)
|
||||
local names = {}
|
||||
for _, a in ipairs(view.decls or {}) do
|
||||
if a.name then names[a.name] = true end
|
||||
end
|
||||
return names
|
||||
end
|
||||
|
||||
local function path_in_module(path, view)
|
||||
if type(path) ~= "string" or path == "" then return false end
|
||||
local norm = path:gsub("\\", "/")
|
||||
local dir = (view.dir or ""):gsub("\\", "/")
|
||||
if dir ~= "" and (norm == dir or norm:sub(1, #dir + 1) == dir .. "/") then
|
||||
return true
|
||||
end
|
||||
for _, src in ipairs(view.sources or {}) do
|
||||
if (src.path or ""):gsub("\\", "/") == norm then return true end
|
||||
end
|
||||
return false
|
||||
end
|
||||
|
||||
local function build_module_view(dir, dir_sources, corpus)
|
||||
local decls = {}
|
||||
for _, src in ipairs(dir_sources or {}) do
|
||||
@@ -352,7 +373,16 @@ local function render_section_reguse(add, view)
|
||||
end
|
||||
add("")
|
||||
end
|
||||
local errors = (view.corpus and view.corpus.reg_use_errors) or {}
|
||||
local bound = {}
|
||||
for _, schema in ipairs(view.schemas or {}) do
|
||||
if schema.name then bound[schema.name] = true end
|
||||
end
|
||||
local errors = {}
|
||||
for _, err in ipairs((view.corpus and view.corpus.reg_use_errors) or {}) do
|
||||
if bound[err.schema_name] or path_in_module(err.source_file, view) then
|
||||
errors[#errors + 1] = err
|
||||
end
|
||||
end
|
||||
if #errors > 0 then
|
||||
wrote = true
|
||||
add("### parse errors")
|
||||
@@ -389,17 +419,36 @@ local function render_section_annotations(add, view)
|
||||
add("")
|
||||
end
|
||||
|
||||
local function render_section_component_annotations(add, view)
|
||||
local rows = {}
|
||||
for _, src in ipairs(view.sources) do
|
||||
for _, info in ipairs((src.scan and src.scan.component_atom_infos) or {}) do
|
||||
rows[#rows + 1] = {
|
||||
source = source_basename(src.path),
|
||||
line = info.info_line or 0,
|
||||
name = info.atom_name or "?",
|
||||
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "—",
|
||||
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "—",
|
||||
}
|
||||
end
|
||||
end
|
||||
if #rows == 0 then add("_(none)_"); add(""); return end
|
||||
add("| source | line | name | reads | writes |")
|
||||
add("|--------|------|------|-------|--------|")
|
||||
for _, r in ipairs(rows) do
|
||||
add(string.format("| %s | %d | %s | %s | %s |",
|
||||
r.source, r.line, r.name, r.reads, r.writes))
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
local function render_section_binds(add, view)
|
||||
local wrote = false
|
||||
for _, src in ipairs(view.sources) do
|
||||
for _, b in ipairs((src.scan and src.scan.binds) or {}) do
|
||||
wrote = true
|
||||
local line = b.line or 0
|
||||
if src.scan.line_of and type(b.line) == "number" then
|
||||
line = src.scan.line_of(b.line) or b.line
|
||||
end
|
||||
add(string.format("### %s (%s:%s, %s bytes)",
|
||||
b.name, source_basename(src.path), tostring(line), tostring(b.bytes or "—")))
|
||||
b.name, source_basename(src.path), tostring(b.line or 0), tostring(b.bytes or "—")))
|
||||
for _, f in ipairs(b.fields or {}) do
|
||||
add(string.format("- `+%s %s`", tostring(f.offset or "?"), f.name or "?"))
|
||||
end
|
||||
@@ -411,46 +460,75 @@ end
|
||||
|
||||
local function render_section_phases(add, view)
|
||||
local corpus = view.corpus or {}
|
||||
local names = decl_names(view)
|
||||
local wrote = false
|
||||
for phase, entry in pairs(corpus.atom_phases or {}) do
|
||||
local here = {}
|
||||
for _, atom_name in ipairs(entry.atoms or {}) do
|
||||
if names[atom_name] then here[#here + 1] = atom_name end
|
||||
end
|
||||
if #here > 0 then
|
||||
wrote = true
|
||||
add(string.format("- phase `%s`: %s", phase, table.concat(entry.atoms or {}, ", ")))
|
||||
add(string.format("- phase `%s`: %s", phase, table.concat(here, ", ")))
|
||||
end
|
||||
end
|
||||
for name, entry in pairs(corpus.atom_views or {}) do
|
||||
if names[name] then
|
||||
wrote = true
|
||||
add(string.format("- view `%s` binds `%s`", name, entry.binds_name or "—"))
|
||||
end
|
||||
end
|
||||
for name, entry in pairs(corpus.atom_ctxs or {}) do
|
||||
if names[name] then
|
||||
wrote = true
|
||||
add(string.format("- ctx `%s` rbind `%s`", name, entry.rbind_atom or "—"))
|
||||
end
|
||||
end
|
||||
if not wrote then add("_(none)_") end
|
||||
add("")
|
||||
end
|
||||
|
||||
local function render_section_aliases(add, view)
|
||||
local reg = (view.corpus and view.corpus.register_alias_registry) or {}
|
||||
local names = {}
|
||||
for name in pairs(reg) do names[#names + 1] = name end
|
||||
local seen = {}
|
||||
for _, src in ipairs(view.sources or {}) do
|
||||
for name, entry in pairs((src.scan and src.scan.register_alias_registry) or {}) do
|
||||
if not seen[name] then
|
||||
seen[name] = entry
|
||||
names[#names + 1] = name
|
||||
end
|
||||
end
|
||||
end
|
||||
table.sort(names)
|
||||
if #names == 0 then add("_(none)_"); add(""); return end
|
||||
add("| alias | type |")
|
||||
add("|-------|------|")
|
||||
for _, name in ipairs(names) do
|
||||
local e = reg[name]
|
||||
local e = seen[name]
|
||||
add(string.format("| %s | %s |", name, (e and e.default_type) or "—"))
|
||||
end
|
||||
add("")
|
||||
end
|
||||
|
||||
local function render_section_autoreg(add, view)
|
||||
local corpus = view.corpus or {}
|
||||
local allowed = decl_names(view)
|
||||
for phase, entry in pairs((view.corpus and view.corpus.atom_phases) or {}) do
|
||||
for _, atom_name in ipairs(entry.atoms or {}) do
|
||||
if allowed[atom_name] then allowed[phase] = true end
|
||||
end
|
||||
end
|
||||
local wrote = false
|
||||
local seen = {}
|
||||
local function dump(label, table_map)
|
||||
local scopes = {}
|
||||
for scope in pairs(table_map or {}) do scopes[#scopes + 1] = scope end
|
||||
for scope in pairs(table_map or {}) do
|
||||
if allowed[scope] and not seen[label .. "\0" .. scope] then
|
||||
scopes[#scopes + 1] = scope
|
||||
end
|
||||
end
|
||||
table.sort(scopes)
|
||||
for _, scope in ipairs(scopes) do
|
||||
seen[label .. "\0" .. scope] = true
|
||||
wrote = true
|
||||
local syms = {}
|
||||
for sym, gpr in pairs(table_map[scope] or {}) do
|
||||
@@ -464,16 +542,28 @@ local function render_section_autoreg(add, view)
|
||||
add(string.format("- %s `%s`: %s", label, scope, table.concat(syms, ", ")))
|
||||
end
|
||||
end
|
||||
local corpus = view.corpus or {}
|
||||
dump("atom", corpus.atom_auto_regs)
|
||||
dump("phase", corpus.phase_auto_regs)
|
||||
for _, src in ipairs(view.sources or {}) do
|
||||
dump("atom", src.scan and src.scan.atom_auto_regs)
|
||||
dump("phase", src.scan and src.scan.phase_auto_regs)
|
||||
end
|
||||
if not wrote then add("_(none)_") end
|
||||
add("")
|
||||
end
|
||||
|
||||
local function render_section_collisions(add, view)
|
||||
local cols = (view.corpus and view.corpus.collisions) or {}
|
||||
if #cols == 0 then add("_(none)_"); add(""); return end
|
||||
for _, c in ipairs(cols) do
|
||||
local rows = {}
|
||||
for _, c in ipairs((view.corpus and view.corpus.collisions) or {}) do
|
||||
local first = c.first_site or {}
|
||||
local other = c.conflicting_site or {}
|
||||
if path_in_module(first.path, view) or path_in_module(other.path, view) then
|
||||
rows[#rows + 1] = c
|
||||
end
|
||||
end
|
||||
if #rows == 0 then add("_(none)_"); add(""); return end
|
||||
for _, c in ipairs(rows) do
|
||||
local first = c.first_site or {}
|
||||
local other = c.conflicting_site or {}
|
||||
add(string.format("- `%s` `%s` first %s:%s conflict %s:%s",
|
||||
@@ -543,19 +633,123 @@ local function render_section_relations(add, view)
|
||||
if not wrote then add("_(none)_"); add("") end
|
||||
end
|
||||
|
||||
local HIDDEN_UNLESS_WRITTEN = {
|
||||
R_AT = true, R_TapePtr = true, R_AtomJmp = true,
|
||||
}
|
||||
|
||||
local PHYSICAL_GPR = {
|
||||
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
|
||||
R_T4 = true, R_T5 = true, R_T6 = true, R_T7 = true,
|
||||
R_V0 = true, R_V1 = true,
|
||||
}
|
||||
|
||||
local function encoder_wrote_key(atom, key)
|
||||
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
|
||||
for _, dest in pairs(ev.gpr_keys or {}) do
|
||||
if dest == key then return true end
|
||||
end
|
||||
end
|
||||
return false
|
||||
end
|
||||
|
||||
local function written_name_for(key, atom)
|
||||
local slot = key:match("^reguse:.+:(.+)$")
|
||||
if slot then
|
||||
local param = atom.reg_use_param_name
|
||||
if param and param ~= "" then return param .. "." .. slot end
|
||||
return slot
|
||||
end
|
||||
return key
|
||||
end
|
||||
|
||||
local function aliases_for_key(key, atom, view)
|
||||
local slot = key:match("^reguse:.+:(.+)$")
|
||||
if not slot then return "—" end
|
||||
local schema_name = atom.reg_use_schema_name
|
||||
local schema = view.corpus and view.corpus.reg_use_schemas and view.corpus.reg_use_schemas[schema_name]
|
||||
if not schema then return "—" end
|
||||
for _, s in ipairs(schema.slots or {}) do
|
||||
if s.name == slot then
|
||||
local names = {}
|
||||
for _, alias in ipairs(s.aliases or {}) do
|
||||
if alias ~= slot then names[#names + 1] = alias end
|
||||
end
|
||||
if #names == 0 then
|
||||
if s.aliases and #s.aliases > 0 then return table.concat(s.aliases, ", ") end
|
||||
return "—"
|
||||
end
|
||||
return table.concat(names, ", ")
|
||||
end
|
||||
end
|
||||
return "—"
|
||||
end
|
||||
|
||||
local function physical_for_key(key, atom, view)
|
||||
if PHYSICAL_GPR[key] then return key end
|
||||
local corpus = view.corpus or {}
|
||||
local alias = (corpus.register_alias_registry or {})[key]
|
||||
if type(alias) == "table" then
|
||||
local phys = alias.physical or alias.gpr or alias.code_name
|
||||
if type(phys) == "string" and PHYSICAL_GPR[phys] then return phys end
|
||||
if type(alias.name) == "string" and PHYSICAL_GPR[alias.name] then return alias.name end
|
||||
elseif type(alias) == "string" and PHYSICAL_GPR[alias] then
|
||||
return alias
|
||||
end
|
||||
local atom_map = (corpus.atom_auto_regs or {})[atom.name]
|
||||
if type(atom_map) == "table" then
|
||||
local slot = key:match("^reguse:.+:(.+)$") or key
|
||||
local bound = atom_map[slot] or atom_map["R_" .. slot]
|
||||
if type(bound) == "string" and PHYSICAL_GPR[bound] then return bound end
|
||||
end
|
||||
return "—"
|
||||
end
|
||||
|
||||
local function last_relation_for(key, atom)
|
||||
local last = nil
|
||||
for _, rel in ipairs((atom.paths and atom.paths.relations) or {}) do
|
||||
local dest = rel.destination or rel.producer_destination
|
||||
if dest == key then last = rel end
|
||||
end
|
||||
if not last then return "—" end
|
||||
local sem = last.semantic or "?"
|
||||
local a = last.producer_word
|
||||
local b = last.consumer_word
|
||||
if a and b then return string.format("%s w%s→%s", sem, tostring(a), tostring(b)) end
|
||||
return sem
|
||||
end
|
||||
|
||||
local function render_section_forward(add, view)
|
||||
local wrote = false
|
||||
for _, a in ipairs(view.decls) do
|
||||
local gpr = a.paths and a.paths.forward_state and a.paths.forward_state.gpr_values
|
||||
if gpr and next(gpr) ~= nil then
|
||||
local keys = {}
|
||||
for k in pairs(gpr or {}) do
|
||||
if k == "R_0" then
|
||||
-- hidden
|
||||
elseif HIDDEN_UNLESS_WRITTEN[k] and not encoder_wrote_key(a, k) then
|
||||
-- hidden
|
||||
else
|
||||
keys[#keys + 1] = k
|
||||
end
|
||||
end
|
||||
if #keys > 0 then
|
||||
wrote = true
|
||||
add("### " .. a.name)
|
||||
local keys = {}
|
||||
for k in pairs(gpr) do keys[#keys + 1] = k end
|
||||
add("| written | aliases | physical | lattice | last relation |")
|
||||
add("|---|---|---|---|---|")
|
||||
table.sort(keys)
|
||||
for _, k in ipairs(keys) do
|
||||
local slot = gpr[k]
|
||||
add(string.format("- `%s` %s", k, (slot and slot.kind) or "unknown"))
|
||||
local lattice = "—"
|
||||
if slot and slot.kind == "constant" then
|
||||
lattice = tostring(slot.value)
|
||||
end
|
||||
add(string.format("| `%s` | %s | %s | %s | %s |",
|
||||
written_name_for(k, a),
|
||||
aliases_for_key(k, a, view),
|
||||
physical_for_key(k, a, view),
|
||||
lattice,
|
||||
last_relation_for(k, a)))
|
||||
end
|
||||
add("")
|
||||
end
|
||||
@@ -568,6 +762,7 @@ local SECTION_RENDERERS = {
|
||||
{ header = "## Components", render = render_section_components },
|
||||
{ header = "## RegUse schemas", render = render_section_reguse },
|
||||
{ header = "## Annotations", render = render_section_annotations },
|
||||
{ header = "## Component annotations", render = render_section_component_annotations },
|
||||
{ header = "## Binds_* structs", render = render_section_binds },
|
||||
{ header = "## Phases / views / ctx", render = render_section_phases },
|
||||
{ header = "## Register aliases", render = render_section_aliases },
|
||||
@@ -575,7 +770,7 @@ local SECTION_RENDERERS = {
|
||||
{ header = "## Collisions", render = render_section_collisions },
|
||||
{ header = "## Findings", render = render_section_findings },
|
||||
{ header = "## Relations", render = render_section_relations },
|
||||
{ header = "## Forward GPR", render = render_section_forward },
|
||||
{ header = "## GPR model", render = render_section_forward },
|
||||
}
|
||||
|
||||
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
|
||||
|
||||
+331
-57
@@ -416,31 +416,51 @@ local function walk_body_fields(body, build_field)
|
||||
return fields
|
||||
end
|
||||
|
||||
-- Parse the `<type> <field>;` declarations from a Struct_ body.
|
||||
-- Parse the `<type> <field>[, <field>...];` declarations from a Struct_ body.
|
||||
-- After the type and `*` chain, keep reading `, ident` until `;`.
|
||||
-- Same type, same pointer depth for every name on that list.
|
||||
-- Returns the raw fields array with `{name, type_name, pointer_depth}` only (NO offset / byte_size).
|
||||
-- The propagation pass `resolve_struct_field_sizes` walks each struct's fields AFTER type resolution and populates offset + byte_size in place.
|
||||
-- Returns (fields). The aggregate byte_count is computed in the propagation pass (it depends on whether every field's type resolved).
|
||||
local function parse_struct_body_fields(body)
|
||||
return walk_body_fields(body, function(type_name, type_end, after_type)
|
||||
-- Parse the trailing `*` chain to derive pointer_depth.
|
||||
local depth, cursor = 0, after_type
|
||||
while cursor <= #body and body:sub(cursor, cursor) == "*" do
|
||||
local fields = {}
|
||||
local body_pos = 1
|
||||
local body_len = #body
|
||||
while body_pos <= body_len do
|
||||
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
|
||||
if body_pos > body_len then break end
|
||||
local type_name, type_end = duffle.read_ident(body, body_pos)
|
||||
if not type_name then
|
||||
body_pos = body_pos + 1
|
||||
else
|
||||
local depth, cursor = 0, duffle.skip_ws_and_cmt(body, type_end)
|
||||
while cursor <= body_len and body:sub(cursor, cursor) == "*" do
|
||||
depth = depth + 1
|
||||
cursor = cursor + 1
|
||||
cursor = duffle.skip_ws_and_cmt(body, cursor)
|
||||
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
|
||||
end
|
||||
-- Read the field ident immediately after the type chain.
|
||||
while cursor <= body_len do
|
||||
local field_ident, field_end = duffle.read_ident(body, cursor)
|
||||
if not field_ident then return nil, type_end + 1 end
|
||||
return {
|
||||
if not field_ident then break end
|
||||
fields[#fields + 1] = {
|
||||
name = field_ident,
|
||||
type_name = type_name,
|
||||
pointer_depth = depth,
|
||||
-- offset + byte_size filled by resolve_struct_field_sizes
|
||||
offset = nil,
|
||||
byte_size = nil,
|
||||
}, field_end
|
||||
end)
|
||||
}
|
||||
cursor = duffle.skip_ws_and_cmt(body, field_end)
|
||||
if body:sub(cursor, cursor) == "," then
|
||||
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
|
||||
else
|
||||
break
|
||||
end
|
||||
end
|
||||
if cursor <= body_len and body:sub(cursor, cursor) == ";" then
|
||||
cursor = cursor + 1
|
||||
end
|
||||
body_pos = cursor
|
||||
end
|
||||
end
|
||||
return fields
|
||||
end
|
||||
|
||||
-- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries.
|
||||
@@ -1252,31 +1272,20 @@ local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
|
||||
return after_paren
|
||||
end
|
||||
|
||||
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @return integer
|
||||
local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||
if not inner then return after_paren end
|
||||
|
||||
local raw_name = duffle.read_ident(inner, 1)
|
||||
|
||||
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
|
||||
local brace_search_pos = after_paren
|
||||
--- Lookahead for `atom_info(...)` after a declaration's closing paren.
|
||||
--- Records into `dest` (atom_infos or component_atom_infos). Returns the position after the info, or after_paren if none.
|
||||
local function parse_atom_info_after_decl(source, after_paren, raw_name, line_of, out, dest)
|
||||
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
|
||||
local look_ident, look_end = duffle.read_ident(source, lookahead)
|
||||
if look_ident == "atom_info" then
|
||||
if look_ident ~= "atom_info" then return after_paren end
|
||||
local info_open = duffle.skip_ws_and_cmt(source, look_end)
|
||||
if source:sub(info_open, info_open) == "(" then
|
||||
if source:sub(info_open, info_open) ~= "(" then return after_paren end
|
||||
local info_inner, info_after = duffle.read_parens(source, info_open)
|
||||
-- info_line feeds the per-atom reg_type_overrides table.
|
||||
if not info_inner then return after_paren end
|
||||
local info_line = line_of(info_open)
|
||||
local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line)
|
||||
out.atom_infos[#out.atom_infos + 1] = {
|
||||
dest = dest or out.atom_infos
|
||||
dest[#dest + 1] = {
|
||||
atom_name = raw_name or "?", binds = ai_binds,
|
||||
reads = ai_reads or {}, writes = ai_writes or {},
|
||||
view = ai_view,
|
||||
@@ -1293,11 +1302,9 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
||||
info_line = line_of(lookahead),
|
||||
}
|
||||
elseif raw_name and ai_overrides then
|
||||
-- Record per-atom overrides even without atom_view.
|
||||
out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) }
|
||||
out.atom_views[raw_name].reg_type_overrides = ai_overrides
|
||||
end
|
||||
-- Project the per-atom atom_ctx / atom_phase declarations onto the global phase index.
|
||||
if raw_name then
|
||||
if ai_ctx then
|
||||
out.atom_ctxs = out.atom_ctxs or {}
|
||||
@@ -1309,10 +1316,25 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
||||
out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name
|
||||
end
|
||||
end
|
||||
brace_search_pos = info_after
|
||||
end
|
||||
return info_after
|
||||
end
|
||||
|
||||
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
--- @param line_of fun(pos: integer): integer
|
||||
--- @param out SourceScan
|
||||
--- @return integer
|
||||
local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||
if not inner then return after_paren end
|
||||
|
||||
local raw_name = duffle.read_ident(inner, 1)
|
||||
|
||||
-- Lookahead for atom_info(...) between `)` and `{`.
|
||||
local brace_search_pos = parse_atom_info_after_decl(source, after_paren, raw_name, line_of, out, out.atom_infos)
|
||||
|
||||
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
|
||||
if not body then return after_brace end
|
||||
if raw_name and raw_name ~= "" then
|
||||
@@ -1336,7 +1358,9 @@ local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
|
||||
local raw_name = duffle.read_ident(inner, 1)
|
||||
if not raw_name then return open_paren + 1 end
|
||||
|
||||
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
|
||||
out.component_atom_infos = out.component_atom_infos or {}
|
||||
local brace_search_pos = parse_atom_info_after_decl(source, after_paren, strip_ac_prefix(raw_name), line_of, out, out.component_atom_infos)
|
||||
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
|
||||
if not body then return after_brace end
|
||||
local name = strip_ac_prefix(raw_name)
|
||||
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
||||
@@ -1407,15 +1431,9 @@ local function parse_mips_atom_comp_proc_map(source, pos, ident_end, line_of, ou
|
||||
return after_paren
|
||||
end
|
||||
|
||||
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
|
||||
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
|
||||
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit
|
||||
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
|
||||
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
|
||||
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
|
||||
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
|
||||
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
|
||||
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
|
||||
--- Parse: `MipsAtom_Proc_(aa, { body })` — body is inside the LAST `{` in args.
|
||||
--- Kind is `atom_proc`. The name is the preceding function ident as written.
|
||||
--- Offsets walk this kind. Components do not emit a `mac_*` alias for it.
|
||||
--- @param source string
|
||||
--- @param pos integer
|
||||
--- @param ident_end integer
|
||||
@@ -1439,13 +1457,17 @@ local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
|
||||
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
||||
if close_pos > #inner + 1 then return after_paren end
|
||||
|
||||
-- The atom name is derived from the preceding function declaration
|
||||
-- (`internal MipsAtom* X_proc(...)`), not from the first macro arg (which
|
||||
-- is now `aa`). The backward walk finds the function decl before open_paren
|
||||
-- and strips the `_proc` suffix.
|
||||
local raw_name, args_inner, func_ident = duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
|
||||
-- The atom name is the preceding function ident as written
|
||||
-- (`internal MipsAtom* X(...)`). The first macro arg is the arena.
|
||||
local raw_name, args_inner, func_ident, after_func_paren =
|
||||
duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
|
||||
if not raw_name then raw_name = "?" end
|
||||
local name = strip_ac_prefix(raw_name)
|
||||
if after_func_paren then
|
||||
parse_atom_info_after_decl(source, after_func_paren, name, line_of, out, out.atom_infos)
|
||||
else
|
||||
parse_atom_info_after_decl(source, pos, name, line_of, out, out.atom_infos)
|
||||
end
|
||||
local reg_use_schema_name = nil
|
||||
local reg_use_param_name = nil
|
||||
if args_inner then
|
||||
@@ -1593,7 +1615,35 @@ local function register_typedef_alias(underlying, name, pos, line_of, out)
|
||||
}
|
||||
end
|
||||
|
||||
local function parse_reg_use_schema_body(body)
|
||||
local parse_reg_use_schema_body
|
||||
|
||||
local function fields_for_reg_type(type_name, type_registry)
|
||||
local reg_name = "Reg_" .. type_name
|
||||
local entry = type_registry and type_registry[reg_name]
|
||||
if entry and entry.fields and #entry.fields > 0 then
|
||||
local names = {}
|
||||
for _, field in ipairs(entry.fields) do
|
||||
if field.name then names[#names + 1] = field.name end
|
||||
end
|
||||
if #names > 0 then return names end
|
||||
end
|
||||
if entry and entry.body and parse_reg_use_schema_body then
|
||||
local schema = parse_reg_use_schema_body(entry.body, type_registry)
|
||||
if schema and schema.slots then
|
||||
local names = {}
|
||||
for _, slot in ipairs(schema.slots) do
|
||||
if slot.name then names[#names + 1] = slot.name end
|
||||
end
|
||||
if #names > 0 then return names end
|
||||
end
|
||||
end
|
||||
return nil
|
||||
end
|
||||
|
||||
parse_reg_use_schema_body = function(body, type_registry, opts)
|
||||
opts = opts or {}
|
||||
local require_types = opts.require_types == true
|
||||
local pending = false
|
||||
local slots = {}
|
||||
local alias_to_slot = {}
|
||||
local slot_names = {}
|
||||
@@ -1728,7 +1778,29 @@ local function parse_reg_use_schema_body(body)
|
||||
after_close = duffle.skip_ws_and_cmt(body, after_close)
|
||||
if body:sub(after_close, after_close) == ";" then after_close = after_close + 1 end
|
||||
pos = after_close
|
||||
elseif first == "Reg" then
|
||||
elseif first == "Reg" or first == "Reg_" then
|
||||
local typed_fields = nil
|
||||
if first == "Reg_" then
|
||||
if body:sub(after, after) ~= "(" then
|
||||
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||
return nil, errors
|
||||
end
|
||||
local type_inner, after_paren = duffle.read_parens(body, after)
|
||||
if not type_inner then
|
||||
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||
return nil, errors
|
||||
end
|
||||
local type_ident = duffle.trim(type_inner)
|
||||
typed_fields = fields_for_reg_type(type_ident, type_registry)
|
||||
if not typed_fields then
|
||||
if require_types then
|
||||
errors[#errors + 1] = { kind = "reguse_unknown_reg_type", type_name = type_ident }
|
||||
else
|
||||
pending = true
|
||||
end
|
||||
end
|
||||
after = duffle.skip_ws_and_cmt(body, after_paren)
|
||||
end
|
||||
local readonly = false
|
||||
local maybe_const, maybe_end = duffle.read_ident(body, after)
|
||||
if maybe_const == "const" then
|
||||
@@ -1741,9 +1813,19 @@ local function parse_reg_use_schema_body(body)
|
||||
return nil, errors
|
||||
end
|
||||
for _, n in ipairs(names) do
|
||||
if first == "Reg_" then
|
||||
if typed_fields then
|
||||
for _, field in ipairs(typed_fields) do
|
||||
local path = n .. "." .. field
|
||||
if not add_alias(path, path) then return nil, errors end
|
||||
if not add_slot(path, { path }, readonly) then return nil, errors end
|
||||
end
|
||||
end
|
||||
else
|
||||
if not add_alias(n, n) then return nil, errors end
|
||||
if not add_slot(n, { n }, readonly) then return nil, errors end
|
||||
end
|
||||
end
|
||||
pos = new_pos
|
||||
else
|
||||
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||
@@ -1752,10 +1834,15 @@ local function parse_reg_use_schema_body(body)
|
||||
::continue::
|
||||
end
|
||||
if #slots == 0 then
|
||||
if pending and not require_types then
|
||||
return { slots = slots, alias_to_slot = alias_to_slot, pending = true }, errors
|
||||
end
|
||||
if #errors == 0 then
|
||||
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||
end
|
||||
return nil, errors
|
||||
end
|
||||
return { slots = slots, alias_to_slot = alias_to_slot }, errors
|
||||
return { slots = slots, alias_to_slot = alias_to_slot, pending = pending }, errors
|
||||
end
|
||||
|
||||
--- Parse: `typedef` declarations.
|
||||
@@ -1792,7 +1879,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
||||
if not body then return after_brace end
|
||||
register_struct_type(body, name, pos, line_of, out)
|
||||
if name:sub(1, 7) == "RegUse_" then
|
||||
local schema, schema_errors = parse_reg_use_schema_body(body)
|
||||
local schema, schema_errors = parse_reg_use_schema_body(body, out.type_name_registry)
|
||||
if schema then
|
||||
schema.name = name
|
||||
schema.source_file = out._source_file
|
||||
@@ -2156,6 +2243,137 @@ local DECL_PARSERS = {
|
||||
-- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`.
|
||||
-- Unknown identifiers follow the same unrelated-token path as every other unsupported source token.
|
||||
|
||||
local TAPE_SKIP_MACROS = {
|
||||
MipsAtom_ = true,
|
||||
MipsAtom_Proc_ = true,
|
||||
MipsAtomComp_ = true,
|
||||
MipsAtomComp_Proc_ = true,
|
||||
MipsAtomComp_ProcMap_ = true,
|
||||
Struct_ = true,
|
||||
Enum_ = true,
|
||||
}
|
||||
|
||||
local function collect_addrs_assigns(text)
|
||||
local addrs = {}
|
||||
local pos = 1
|
||||
local n = #text
|
||||
while pos <= n do
|
||||
pos = duffle.skip_ws_and_cmt(text, pos)
|
||||
if pos > n then break end
|
||||
local ident, ident_end = duffle.read_ident(text, pos)
|
||||
if ident == "addrs" then
|
||||
local after = duffle.skip_ws_and_cmt(text, ident_end)
|
||||
if text:sub(after, after) == "[" then
|
||||
local inner, after_br = duffle.read_brackets(text, after)
|
||||
local idx = inner and tonumber(duffle.trim(inner))
|
||||
after_br = duffle.skip_ws_and_cmt(text, after_br or after)
|
||||
if idx and text:sub(after_br, after_br) == "=" then
|
||||
local rhs = duffle.skip_ws_and_cmt(text, after_br + 1)
|
||||
local rhs_ident = duffle.read_ident(text, rhs)
|
||||
if rhs_ident then addrs[idx] = rhs_ident end
|
||||
pos = rhs
|
||||
else
|
||||
pos = after_br or (after + 1)
|
||||
end
|
||||
else
|
||||
pos = ident_end
|
||||
end
|
||||
elseif ident then
|
||||
pos = ident_end
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
return addrs
|
||||
end
|
||||
|
||||
local function collect_tb_emits(body, addrs)
|
||||
local names = {}
|
||||
local pos = 1
|
||||
local n = #body
|
||||
while pos <= n do
|
||||
pos = duffle.skip_ws_and_cmt(body, pos)
|
||||
if pos > n then break end
|
||||
local ident, ident_end = duffle.read_ident(body, pos)
|
||||
if ident == "tb_emit_" or ident == "tb_emit" then
|
||||
local after = duffle.skip_ws_and_cmt(body, ident_end)
|
||||
if body:sub(after, after) == "(" then
|
||||
local inner, after_p = duffle.read_parens(body, after)
|
||||
local name
|
||||
if ident == "tb_emit_" then
|
||||
name = duffle.trim(inner or ""):match("^([%w_]+)")
|
||||
else
|
||||
local args = duffle.split_top_level_commas(inner or "")
|
||||
local last = duffle.trim(args[#args] or "")
|
||||
local idx = last:match("^addrs%s*%[%s*(%d+)%s*%]$")
|
||||
if idx then
|
||||
name = addrs[tonumber(idx)]
|
||||
else
|
||||
name = last:match("([%w_]+)$")
|
||||
end
|
||||
end
|
||||
if name then names[#names + 1] = name end
|
||||
pos = after_p or (after + 1)
|
||||
else
|
||||
pos = ident_end
|
||||
end
|
||||
elseif ident then
|
||||
pos = ident_end
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
return names
|
||||
end
|
||||
|
||||
-- Linear appearance order of tb_emit / tb_emit_ in each C function body.
|
||||
-- Commented-out emits are skipped by skip_ws_and_cmt. No C if/loop CFG.
|
||||
local function scan_tape_chains(source)
|
||||
local addrs = collect_addrs_assigns(source)
|
||||
local chains = {}
|
||||
local pos = 1
|
||||
local n = #source
|
||||
while pos <= n do
|
||||
pos = duffle.skip_ws_and_cmt(source, pos)
|
||||
if pos > n then break end
|
||||
local ident, ident_end = duffle.read_ident(source, pos)
|
||||
if ident and TAPE_SKIP_MACROS[ident] then
|
||||
local after = duffle.skip_ws_and_cmt(source, ident_end)
|
||||
if source:sub(after, after) == "(" then
|
||||
local _, after_p = duffle.read_parens(source, after)
|
||||
after = duffle.skip_ws_and_cmt(source, after_p or after)
|
||||
end
|
||||
if source:sub(after, after) == "{" then
|
||||
local _, after_b = duffle.read_braces(source, after)
|
||||
pos = after_b or (after + 1)
|
||||
else
|
||||
pos = after
|
||||
end
|
||||
elseif ident then
|
||||
local after = duffle.skip_ws_and_cmt(source, ident_end)
|
||||
if source:sub(after, after) == "(" then
|
||||
local _, after_p = duffle.read_parens(source, after)
|
||||
after = duffle.skip_ws_and_cmt(source, after_p or after)
|
||||
if source:sub(after, after) == "{" then
|
||||
local body, after_b = duffle.read_braces(source, after)
|
||||
local names = collect_tb_emits(body or "", addrs)
|
||||
if #names > 0 then
|
||||
chains[#chains + 1] = names
|
||||
end
|
||||
pos = after_b or (after + 1)
|
||||
else
|
||||
pos = after
|
||||
end
|
||||
else
|
||||
pos = ident_end
|
||||
end
|
||||
else
|
||||
pos = pos + 1
|
||||
end
|
||||
end
|
||||
return chains
|
||||
end
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
-- The single source walker
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -2174,6 +2392,7 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
||||
raw_atoms = {},
|
||||
binds = {},
|
||||
atom_infos = {},
|
||||
component_atom_infos = {},
|
||||
macros = {},
|
||||
-- Raw marker evidence for annotation validation. The `debug_skip` boolean
|
||||
-- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip.
|
||||
@@ -2267,6 +2486,7 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
||||
-- Runs AFTER the source walk so all typedef / Struct_ / Enum_ declarations have been parsed into `out.type_name_registry`.
|
||||
-- Mutates each entry's `byte_size` field in place; fields with pointer_depth > 0 already carry byte_size = 4 from parse time and are unaffected.
|
||||
propagate_type_sizes(out)
|
||||
out.tape_chains = scan_tape_chains(source)
|
||||
|
||||
return out
|
||||
end
|
||||
@@ -2429,18 +2649,21 @@ local function merge_corpus_registries(corpus)
|
||||
corpus.atom_ctxs = corpus.atom_ctxs or {}
|
||||
corpus.atom_phases = corpus.atom_phases or {}
|
||||
corpus.atom_infos = corpus.atom_infos or {}
|
||||
corpus.component_atom_infos = corpus.component_atom_infos or {}
|
||||
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
|
||||
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
|
||||
corpus.collisions = corpus.collisions or {}
|
||||
corpus.reg_use_schemas = corpus.reg_use_schemas or {}
|
||||
corpus.reg_use_errors = corpus.reg_use_errors or {}
|
||||
corpus.tape_chains = corpus.tape_chains or {}
|
||||
|
||||
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
||||
-- This is safe because M.run is the only writer to these tables within a single orchestrator invocation.
|
||||
for _, key in ipairs({
|
||||
"register_alias_registry", "type_name_registry", "binds_by_name",
|
||||
"atoms_by_name", "atom_views", "atom_ctxs", "atom_phases",
|
||||
"atom_infos", "collisions", "reg_use_schemas", "reg_use_errors",
|
||||
"atom_infos", "component_atom_infos", "collisions", "reg_use_schemas", "reg_use_errors",
|
||||
"tape_chains",
|
||||
}) do
|
||||
corpus[key] = {}
|
||||
end
|
||||
@@ -2533,6 +2756,9 @@ local function merge_corpus_registries(corpus)
|
||||
for _, info in ipairs(scan.atom_infos or {}) do
|
||||
corpus.atom_infos[#corpus.atom_infos + 1] = info
|
||||
end
|
||||
for _, info in ipairs(scan.component_atom_infos or {}) do
|
||||
corpus.component_atom_infos[#corpus.component_atom_infos + 1] = info
|
||||
end
|
||||
|
||||
for name, schema in pairs(scan.reg_use_schemas or {}) do
|
||||
if corpus.reg_use_schemas[name] == nil then
|
||||
@@ -2542,6 +2768,53 @@ local function merge_corpus_registries(corpus)
|
||||
for _, err in ipairs(scan.reg_use_errors or {}) do
|
||||
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
|
||||
end
|
||||
for _, chain in ipairs(scan.tape_chains or {}) do
|
||||
corpus.tape_chains[#corpus.tape_chains + 1] = chain
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local SCHEMA_BODY_ERROR = {
|
||||
reguse_malformed = true,
|
||||
reguse_unknown_reg_type = true,
|
||||
reguse_duplicate_alias = true,
|
||||
reguse_duplicate_slot = true,
|
||||
reguse_const_reg_spelling = true,
|
||||
reguse_mixed_const = true,
|
||||
}
|
||||
|
||||
-- Re-parse every RegUse_* body against the merged type_name_registry.
|
||||
-- Scan-time expansion still runs when Reg_T is in the same source.
|
||||
-- Missing Reg_T after merge is reguse_unknown_reg_type, not a fallback table.
|
||||
local function resolve_reg_use_schemas(corpus)
|
||||
local kept = {}
|
||||
for _, err in ipairs(corpus.reg_use_errors or {}) do
|
||||
if not SCHEMA_BODY_ERROR[err.kind] then
|
||||
kept[#kept + 1] = err
|
||||
end
|
||||
end
|
||||
corpus.reg_use_errors = kept
|
||||
|
||||
for name, type_entry in pairs(corpus.type_name_registry or {}) do
|
||||
if name:sub(1, 7) == "RegUse_" and type_entry.body then
|
||||
local fresh, errs = parse_reg_use_schema_body(
|
||||
type_entry.body, corpus.type_name_registry, { require_types = true })
|
||||
if fresh then
|
||||
fresh.name = name
|
||||
local old = corpus.reg_use_schemas[name]
|
||||
fresh.source_file = (old and old.source_file) or type_entry.source_file
|
||||
fresh.source_line = (old and old.source_line) or type_entry.source_line
|
||||
corpus.reg_use_schemas[name] = fresh
|
||||
else
|
||||
corpus.reg_use_schemas[name] = nil
|
||||
end
|
||||
for _, err in ipairs(errs or {}) do
|
||||
err.schema_name = name
|
||||
err.source_file = type_entry.source_file
|
||||
err.source_line = type_entry.source_line
|
||||
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -2630,6 +2903,7 @@ function M.run(ctx)
|
||||
|
||||
-- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline).
|
||||
merge_corpus_registries(corpus)
|
||||
resolve_reg_use_schemas(corpus)
|
||||
|
||||
-- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return.
|
||||
return { outputs = {}, errors = {}, warnings = {} }
|
||||
|
||||
@@ -279,6 +279,16 @@ local function classify_tokens(tokens)
|
||||
for tok_idx, t in ipairs(tokens) do
|
||||
local tok = t.tok
|
||||
local ident = tok:match("^([%w_]+)") or "?"
|
||||
local is_delay_marker = false
|
||||
local delay_marker = nil
|
||||
if duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[ident] then
|
||||
is_delay_marker = true
|
||||
delay_marker = ident
|
||||
local rest = tok:match("^[%w_]+%s+(.*)$")
|
||||
if rest and rest ~= "" then
|
||||
ident = rest:match("^([%w_]+)") or ident
|
||||
end
|
||||
end
|
||||
local nop_words = 0
|
||||
if ident == "nop" then nop_words = 1
|
||||
elseif ident == "nop2" then nop_words = 2 end
|
||||
@@ -330,7 +340,9 @@ local function classify_tokens(tokens)
|
||||
local shape = ident:match("^mac_format_([%w_]+)_color$")
|
||||
if shape then mac_format_shape = shape end
|
||||
if ident:match("^mac_gte_store_[%w_]+$") then is_gte_store = true end
|
||||
if ident:match("^mac_insert_ot_tag_[%w_]+$") then is_ot_tag = true end
|
||||
if ident == "mac_insert_ot_tag" or ident:match("^mac_insert_ot_tag_[%w_]+$") then
|
||||
is_ot_tag = true
|
||||
end
|
||||
|
||||
-- O_(<arg1>, <arg2>) / S_(<arg>) captures (used by check_abi_handoff).
|
||||
-- Cheap pattern match — anchored, fails fast on non-matching tokens.
|
||||
@@ -343,6 +355,8 @@ local function classify_tokens(tokens)
|
||||
|
||||
tc[tok_idx] = {
|
||||
ident = ident,
|
||||
is_delay_marker = is_delay_marker,
|
||||
delay_marker = delay_marker,
|
||||
nop_words = nop_words,
|
||||
nop_prefix = nop_run,
|
||||
is_yield = is_yield,
|
||||
@@ -961,6 +975,7 @@ local function analyze_hardware_relations(atom)
|
||||
semantic = relation.semantic,
|
||||
producer_word = prod.word,
|
||||
consumer_word = ev_word,
|
||||
destination = prod.destination,
|
||||
gap = gap,
|
||||
required = prod.required,
|
||||
satisfied = satisfied,
|
||||
@@ -1697,9 +1712,10 @@ end
|
||||
--- and the tape runtime would jump to garbage.
|
||||
---
|
||||
--- Rules:
|
||||
--- 1. Every `mac_yield_load()` must be in a branch BD-slot (the immediately preceding token must be a branch).
|
||||
--- 2. Every `mac_yield_tail()` must be the first instruction after an `atom_label()`, AND
|
||||
--- at least one branch targeting that label must have `mac_yield_load()` in its BD-slot.
|
||||
--- 1. Every `mac_yield_load()` must be in a branch BD-slot, or sit between two `atom_label`s.
|
||||
--- Delay-marker prefixes are skipped when reading prev/next tokens.
|
||||
--- 2. `mac_yield_tail()` is valid if every path that reaches it has already executed a `mac_yield_load()`.
|
||||
--- A load in a branch BD slot always runs. Later branches that target the tail label may carry `nop`.
|
||||
--- 3. `mac_yield_tail()` as the atom-end terminator (last token) is a WARNING, not an error
|
||||
--- (the safe default for atom-endings is `mac_yield()` which re-loads `R_AtomJmp`).
|
||||
---
|
||||
@@ -1718,53 +1734,118 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
||||
return atom.line + line_in_body[tokens[idx].rel]
|
||||
end
|
||||
|
||||
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s (natural fall-through load pattern).
|
||||
-- When the pattern is satisfied, the check stays silent; only violations emit findings.
|
||||
local function is_delay_only(c)
|
||||
return c and duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[c.ident] == true
|
||||
end
|
||||
|
||||
local function skip_delay(idx, step)
|
||||
local i = idx
|
||||
while i >= 1 and i <= n and is_delay_only(tc[i]) do
|
||||
i = i + step
|
||||
end
|
||||
if i < 1 or i > n then return nil end
|
||||
return i
|
||||
end
|
||||
|
||||
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s.
|
||||
for tok_idx = 1, n do
|
||||
local c = tc[tok_idx]
|
||||
if c.ident == "mac_yield_load" then
|
||||
local prev_tc = (tok_idx >= 2) and tc[tok_idx - 1] or nil
|
||||
-- Look for the next `atom_label()` token (skip `atom_offset` markers; check immediately-adjacent first).
|
||||
local next_label_tc = (tok_idx + 1 <= n) and tc[tok_idx + 1] or nil
|
||||
if next_label_tc and next_label_tc.ident ~= "atom_label" then
|
||||
next_label_tc = nil
|
||||
for j = tok_idx + 1, n do
|
||||
local prev_i = skip_delay(tok_idx - 1, -1)
|
||||
local prev_tc = prev_i and tc[prev_i] or nil
|
||||
local next_label_tc = nil
|
||||
local j = skip_delay(tok_idx + 1, 1)
|
||||
while j do
|
||||
local t = tc[j]
|
||||
if t.ident == "atom_label" then
|
||||
next_label_tc = t
|
||||
break
|
||||
end
|
||||
end
|
||||
if t.ident ~= "atom_offset" then break end
|
||||
j = skip_delay(j + 1, 1)
|
||||
end
|
||||
local natural_fallthrough = prev_tc and prev_tc.is_atom_label and next_label_tc ~= nil
|
||||
if not natural_fallthrough then
|
||||
if tok_idx < 2 or not prev_tc.is_branch then
|
||||
if not prev_tc or not prev_tc.is_branch then
|
||||
local prev_ident = prev_tc and (prev_tc.ident or "?") or "<none>"
|
||||
local next_ident = next_label_tc and (next_label_tc.ident .. "(" .. (next_label_tc.label_name or "?") .. ")") or "<no following label>"
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
|
||||
line = prev_i and line_for(tok_idx) or atom.line,
|
||||
check = "yield_load_tail_pairing",
|
||||
kind = "error",
|
||||
msg = string.format(
|
||||
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — and the next `atom_label()` token is `%s` — `mac_yield_load()` must fill a branch BD-slot or sit between two `atom_label`s for the natural fall-through load."
|
||||
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
|
||||
, atom.name, prev_i and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- ── Rule 2: every `mac_yield_tail()` must be at a labeled target whose branch BD-slot is `mac_yield_load()`.
|
||||
-- ── Rule 2: `mac_yield_tail()` is valid if every path that reaches it already ran `mac_yield_load()`.
|
||||
-- MIPS delay slot always runs. Successors skip the BD token for control flow,
|
||||
-- but the yield walk still counts that token as executed.
|
||||
local function load_covers_tail(tail_idx)
|
||||
local labels = {}
|
||||
for i = 1, n do
|
||||
if tc[i].is_atom_label and tc[i].label_name then
|
||||
labels[tc[i].label_name] = i
|
||||
end
|
||||
end
|
||||
local function is_load(idx)
|
||||
return tc[idx] and tc[idx].ident == "mac_yield_load"
|
||||
end
|
||||
local reached_without = false
|
||||
local reached_any = false
|
||||
local path_n = 0
|
||||
local MAX_PATHS = 64
|
||||
local function dfs(idx, saw_load, visited)
|
||||
if path_n >= MAX_PATHS then return end
|
||||
if visited[idx] then return end
|
||||
local vis = {}
|
||||
for k, v in pairs(visited) do vis[k] = v end
|
||||
vis[idx] = true
|
||||
local saw = saw_load or is_load(idx)
|
||||
if tc[idx].is_branch and idx + 1 <= n then
|
||||
saw = saw or is_load(idx + 1)
|
||||
end
|
||||
if idx == tail_idx then
|
||||
path_n = path_n + 1
|
||||
reached_any = true
|
||||
if not saw then reached_without = true end
|
||||
return
|
||||
end
|
||||
if tc[idx].is_yield or tc[idx].is_terminal_jump then
|
||||
return
|
||||
end
|
||||
if tc[idx].is_branch then
|
||||
if not tc[idx].is_unconditional_jump and idx + 2 <= n then
|
||||
dfs(idx + 2, saw, vis)
|
||||
end
|
||||
local label = tc[idx].branch_label
|
||||
if label and labels[label] then
|
||||
local dest = labels[label] + 1
|
||||
if dest <= n then dfs(dest, saw, vis) end
|
||||
end
|
||||
return
|
||||
end
|
||||
if idx + 1 <= n then
|
||||
dfs(idx + 1, saw, vis)
|
||||
end
|
||||
end
|
||||
dfs(1, false, {})
|
||||
if not reached_any then return true end
|
||||
return not reached_without
|
||||
end
|
||||
|
||||
for tok_idx = 1, n do
|
||||
local c = tc[tok_idx]
|
||||
if c.ident ~= "mac_yield_tail" then goto continue end
|
||||
|
||||
-- The immediately preceding token must be an `atom_label()` (no instructions between them).
|
||||
local prev_idx = tok_idx - 1
|
||||
if prev_idx < 1 or not tc[prev_idx].is_atom_label then
|
||||
local prev_idx = skip_delay(tok_idx - 1, -1)
|
||||
if not prev_idx or not tc[prev_idx].is_atom_label then
|
||||
if tok_idx == n then
|
||||
-- Atom-ending case: last token is `mac_yield_tail()` without a preceding label. WARNING.
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = line_for(tok_idx),
|
||||
@@ -1789,36 +1870,14 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
||||
end
|
||||
|
||||
local label_name = tc[prev_idx].label_name
|
||||
-- Find at least one branch targeting `label_name` whose BD-slot is `mac_yield_load()`.
|
||||
local found_pairing = false
|
||||
for branch_idx = 1, n do
|
||||
local bt = tc[branch_idx]
|
||||
if bt.is_branch and bt.branch_label == label_name then
|
||||
local bd_idx = branch_idx + 1
|
||||
local bd_tc = bd_idx <= n and tc[bd_idx] or nil
|
||||
if bd_tc and bd_tc.ident == "mac_yield_load" then
|
||||
found_pairing = true
|
||||
else
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = line_for(branch_idx),
|
||||
check = "yield_load_tail_pairing",
|
||||
kind = "error",
|
||||
msg = string.format(
|
||||
"%s at line %d has `mac_yield_tail()` at label `%s` (word %d) but the branch targeting it (at word %d) has BD-slot `%s` instead of `mac_yield_load()`."
|
||||
, atom.name, line_for(branch_idx), label_name, tok_idx, branch_idx, bd_tc and bd_tc.ident or "?"),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
if not found_pairing then
|
||||
if not load_covers_tail(tok_idx) then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = line_for(tok_idx),
|
||||
check = "yield_load_tail_pairing",
|
||||
kind = "error",
|
||||
msg = string.format(
|
||||
"%s at line %d has `mac_yield_tail()` at label `%s` but no branch in the body targets this label with `mac_yield_load()` in its BD-slot — R_AtomJmp would not be loaded."
|
||||
"%s at line %d has `mac_yield_tail()` at label `%s` but no path that reaches it has executed `mac_yield_load()` — R_AtomJmp would not be loaded."
|
||||
, atom.name, line_for(tok_idx), label_name),
|
||||
}
|
||||
end
|
||||
@@ -1936,6 +1995,7 @@ local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
|
||||
local contrib = 0
|
||||
local saw_format = false
|
||||
local saw_prim_write = false
|
||||
local saw_tag = false
|
||||
|
||||
-- Reads from tc_entry fields pre-computed by classify_tokens (R3 lift).
|
||||
-- Eliminates 4 per-token string matches (mac_format_X_color + mac_gte_store_<shape> + mac_insert_ot_tag_<shape> + R_PrimCursor)
|
||||
@@ -1966,10 +2026,57 @@ local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
|
||||
local comp = pipe_ctx.components_by_name[bare]
|
||||
local n = comp and comp.gp0_contrib
|
||||
if n then contrib = contrib + n end
|
||||
-- insert_ot_tag writes the packet tag. Count it once when the atom
|
||||
-- has no raw O_(Poly_*, tag) store.
|
||||
if not saw_tag then
|
||||
contrib = contrib + 1
|
||||
saw_tag = true
|
||||
end
|
||||
end
|
||||
if tc_entry.writes_r_prim_cursor then
|
||||
saw_prim_write = true
|
||||
end
|
||||
-- A raw store to O_(Poly_*, tag) is the packet tag word, counted once.
|
||||
if tc_entry.o_arg1 and tc_entry.o_arg1:match("^Poly_") and tc_entry.o_arg2 == "tag" then
|
||||
if tc_entry.is_store_word or tc_entry.ident == "gte_sw" then
|
||||
if not saw_tag then
|
||||
contrib = contrib + 1
|
||||
saw_tag = true
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- Token-name gp0_contrib is 0 when bodies use uncounted stores.
|
||||
-- Once gte_sw is taught, prefer that sum. Do not also count every PrimCursor store.
|
||||
if contrib == 0 then
|
||||
local seen_field = {}
|
||||
for _, ev in ipairs(atom.paths.word_events or {}) do
|
||||
local enc = ev.encoder or ""
|
||||
if enc == "store_word" or enc == "store_half" or enc == "store_byte" or enc == "gte_sw" then
|
||||
local text = (ev.call_text or "") .. " " .. (ev.root_call_text or "")
|
||||
if text:find("insert_ot_tag", 1, true) then
|
||||
-- OT list mutation, not a packet word.
|
||||
else
|
||||
local field = text:match("O_%(([^%)]+)%)") or text
|
||||
if not seen_field[field] then
|
||||
local hit = text:find("R_PrimCursor", 1, true)
|
||||
if not hit then
|
||||
for _, arg in ipairs(ev.args or {}) do
|
||||
if tostring(arg):find("R_PrimCursor", 1, true) then
|
||||
hit = true
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
if hit then
|
||||
seen_field[field] = true
|
||||
contrib = contrib + 1
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
if not cmd_byte then
|
||||
@@ -2048,7 +2155,9 @@ local function analyze_atom_paths(atom, pipe_ctx)
|
||||
local c = tc[tok_idx]
|
||||
local ident = c.ident
|
||||
local cost
|
||||
if ident:sub(1, #"mac_") == "mac_" then
|
||||
if duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[ident] then
|
||||
cost = 0
|
||||
elseif ident:sub(1, #"mac_") == "mac_" then
|
||||
-- `mac_*` token: lookup corpus.components[bare_name].cycle_cost.
|
||||
local bare = ident:sub(#"mac_" + 1)
|
||||
local comp = pipe_ctx.components_by_name and pipe_ctx.components_by_name[bare]
|
||||
@@ -2227,13 +2336,11 @@ end
|
||||
-- Per-source rule (called once per source via the CHECK_RULES dispatch).
|
||||
-- Signature matches the per_source shape established by check_semantic_reg_defaults.
|
||||
--
|
||||
-- Severity: WARNING (build continues).
|
||||
-- The rule is intentionally permissive because the production `code/duffle/` and `code/gte_hello/`
|
||||
-- sources use R_* aliases in atom_reads / atom_writes that may not yet be opted in via the bare `atom_reg` marker.
|
||||
-- R_TapePtr / R_AtomJmp / R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase ARE opted in.
|
||||
-- Raw C-ABI aliases like R_T0..R_T3 require explicit opt-in; the prototype keeps wave-context registration explicit.
|
||||
-- no auto-include of wave-context; explicit opt-in only).
|
||||
-- Warnings keep the build green and report aliases that need explicit registration.
|
||||
-- Physical GPRs are members. Opt-in atom_reg stays for aliases.
|
||||
local function is_physical_gpr(reg)
|
||||
return type(reg) == "string" and (reg:match("^R_T[0-7]$") ~= nil or reg:match("^R_V[01]$") ~= nil)
|
||||
end
|
||||
|
||||
local function check_enum_alias_membership(_src, pipe_ctx, findings)
|
||||
local reg_registry = pipe_ctx.register_alias_registry or {}
|
||||
|
||||
@@ -2271,7 +2378,7 @@ local function check_enum_alias_membership(_src, pipe_ctx, findings)
|
||||
end
|
||||
end
|
||||
for _, reg in ipairs(ai.reads or {}) do
|
||||
if not reg_registry[reg] then
|
||||
if not reg_registry[reg] and not is_physical_gpr(reg) then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom_name, line = info_line,
|
||||
check = "enum_alias_membership", kind = "warning",
|
||||
@@ -2281,7 +2388,7 @@ local function check_enum_alias_membership(_src, pipe_ctx, findings)
|
||||
end
|
||||
end
|
||||
for _, reg in ipairs(ai.writes or {}) do
|
||||
if not reg_registry[reg] then
|
||||
if not reg_registry[reg] and not is_physical_gpr(reg) then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom_name, line = info_line,
|
||||
check = "enum_alias_membership", kind = "warning",
|
||||
@@ -2352,6 +2459,18 @@ local function find_field_by_name(type_entry, field_name)
|
||||
return nil
|
||||
end
|
||||
|
||||
-- Walk typedef aliases to the struct that owns the fields table. Depth matches propagate_type_sizes.
|
||||
local function resolve_type_with_fields(type_name, type_registry, depth)
|
||||
if depth > 8 then return nil end
|
||||
local entry = type_registry[type_name]
|
||||
if not entry then return nil end
|
||||
if entry.fields then return entry end
|
||||
if entry.kind == "typedef" and entry.underlying_type and entry.underlying_type ~= "" then
|
||||
return resolve_type_with_fields(entry.underlying_type, type_registry, depth + 1)
|
||||
end
|
||||
return entry
|
||||
end
|
||||
|
||||
-- True iff a (field, type_registry) pair is a leaf scalar (safe to dereference as a tape-payload field).
|
||||
-- Pointer-to-X is always a leaf; non-pointer struct members fail the leaf test.
|
||||
local function is_field_leaf(field, type_registry)
|
||||
@@ -2379,8 +2498,18 @@ local function check_binds_no_substruct_deref(_src, pipe_ctx, findings)
|
||||
local field_name = tc_entry.o_arg2
|
||||
local body_line = a.line + (line_in_body[tokens[ti].rel] or 0)
|
||||
|
||||
local type_entry = type_registry[type_name]
|
||||
if not type_entry or not type_entry.fields then
|
||||
local type_entry = resolve_type_with_fields(type_name, type_registry, 1)
|
||||
local no_fields = not type_entry or not type_entry.fields or #type_entry.fields == 0
|
||||
local raw_entry = type_registry[type_name]
|
||||
local is_typedef_to_struct = raw_entry
|
||||
and raw_entry.kind == "typedef"
|
||||
and raw_entry.underlying_type
|
||||
and type_registry[raw_entry.underlying_type]
|
||||
and type_registry[raw_entry.underlying_type].fields
|
||||
local skip_opaque = no_fields and not is_typedef_to_struct
|
||||
if skip_opaque then
|
||||
-- leave this token
|
||||
elseif not type_entry or not type_entry.fields then
|
||||
findings[#findings + 1] = {
|
||||
atom = a.name, line = body_line,
|
||||
check = "binds_no_substruct_deref", kind = "warning",
|
||||
@@ -2450,6 +2579,44 @@ local function atom_body_token_source_line(atom, token, line_in_body)
|
||||
return (atom.line or 0) + body_line - 1
|
||||
end
|
||||
|
||||
local function ctrl_alias_from_text(text)
|
||||
return tostring(text or ""):match("gte_cr_[%w_]+")
|
||||
end
|
||||
|
||||
local function ctrl_writes_in_atom(atom)
|
||||
local out = {}
|
||||
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
|
||||
if (ev.encoder or "") == "gte_mv_to_ctrl_r" then
|
||||
local alias = ev.args and ev.args[2]
|
||||
if type(alias) ~= "string" or not alias:match("^gte_cr_") then
|
||||
alias = ctrl_alias_from_text(ev.call_text) or ctrl_alias_from_text(ev.root_call_text)
|
||||
end
|
||||
local src = ev.args and ev.args[1]
|
||||
if type(src) == "string" then src = src:match("[%w_]+") end
|
||||
if alias then
|
||||
out[#out + 1] = {
|
||||
alias = alias,
|
||||
src = src,
|
||||
line = ev.line or atom.line,
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
if #out == 0 then
|
||||
for _, t in ipairs((atom.paths and atom.paths.tokens) or {}) do
|
||||
local tok = t.tok or ""
|
||||
if tok:match("^gte_mv_to_ctrl_r") then
|
||||
local alias = ctrl_alias_from_text(tok)
|
||||
local src = tok:match("%(%s*([%w_]+)")
|
||||
if alias then
|
||||
out[#out + 1] = { alias = alias, src = src, line = atom.line }
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
|
||||
-- Check #N: gte_cr_alias_writes
|
||||
-- Fires one warning per atom per alias-group when the atom body touches two
|
||||
-- distinct aliases from the same group. Aliases within a group write to the
|
||||
@@ -2590,6 +2757,173 @@ local function check_gte_cr_TR_naming(atom, _pipe_ctx, findings)
|
||||
end
|
||||
end
|
||||
|
||||
local function check_gte_cr_alias_writes_xatom(src, pipe_ctx, findings)
|
||||
-- Walk tape chains once (first source only). Atoms in no chain stay per-atom.
|
||||
local first = pipe_ctx.source_order and pipe_ctx.source_order[1]
|
||||
if first and src ~= first then return end
|
||||
local atoms_by_name = pipe_ctx.atoms_by_name or {}
|
||||
for _, chain in ipairs(pipe_ctx.tape_chains or {}) do
|
||||
local slot_state = {}
|
||||
for _, name in ipairs(chain) do
|
||||
local atom = atoms_by_name[name]
|
||||
if atom then
|
||||
atom.paths = atom.paths or {}
|
||||
atom.paths.forward_state = atom.paths.forward_state or {}
|
||||
local outgoing = {}
|
||||
for slot, prev in pairs(slot_state) do
|
||||
outgoing[slot] = prev
|
||||
end
|
||||
for _, w in ipairs(ctrl_writes_in_atom(atom)) do
|
||||
local group = find_alias_pair_for(w.alias, duffle)
|
||||
if group then
|
||||
local slot = group[1]
|
||||
local prev = slot_state[slot]
|
||||
if prev and prev.alias ~= w.alias and prev.atom ~= atom.name then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name or "",
|
||||
line = w.line,
|
||||
check = "gte_cr_alias_writes_xatom",
|
||||
kind = "warning",
|
||||
msg = string.format(
|
||||
"atom '%s' writes %s to C2[%d]; atom '%s' already wrote %s"
|
||||
, atom.name or "", w.alias, slot, prev.atom, prev.alias),
|
||||
}
|
||||
end
|
||||
slot_state[slot] = { alias = w.alias, atom = atom.name, line = w.line }
|
||||
outgoing[slot] = slot_state[slot]
|
||||
end
|
||||
end
|
||||
atom.paths.forward_state.ctrl_writes_by_slot = outgoing
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local function check_gte_packed_writes(atom, _pipe_ctx, findings)
|
||||
local writes = ctrl_writes_in_atom(atom)
|
||||
local first_idx = {}
|
||||
for i, w in ipairs(writes) do
|
||||
if first_idx[w.alias] == nil then first_idx[w.alias] = i end
|
||||
end
|
||||
for _, rel in ipairs(duffle.GTE_PACKED_SLOT_RELATIONS or {}) do
|
||||
local i1 = first_idx[rel.first]
|
||||
local i2 = first_idx[rel.second]
|
||||
if i1 and i2 and i2 < i1 then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name or "",
|
||||
line = writes[i2].line,
|
||||
check = "gte_packed_writes",
|
||||
kind = "warning",
|
||||
msg = string.format(
|
||||
"atom '%s' writes %s before %s on packed C2[%d]"
|
||||
, atom.name or "", rel.second, rel.first, rel.slot),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local function check_ctc2_chain_source_preservation(atom, _pipe_ctx, findings)
|
||||
-- Fire only when a load sits before a later RT ctc2 that still names that GPR.
|
||||
-- A load after the last RT ctc2 and before the command is a legal reload.
|
||||
local events = (atom.paths and atom.paths.word_events) or {}
|
||||
local function event_src(ev)
|
||||
if ev.gpr_keys and ev.gpr_keys[1] then return ev.gpr_keys[1] end
|
||||
local src = ev.args and ev.args[1]
|
||||
if type(src) == "string" then src = src:match("^[%w_.]+") end
|
||||
return src
|
||||
end
|
||||
local function event_alias(ev)
|
||||
local alias = ev.args and ev.args[2]
|
||||
if type(alias) ~= "string" or not alias:match("^gte_cr_") then
|
||||
alias = ctrl_alias_from_text(ev.call_text)
|
||||
end
|
||||
return alias
|
||||
end
|
||||
if #events > 0 then
|
||||
for i, ev in ipairs(events) do
|
||||
local enc = ev.encoder or ""
|
||||
if enc == "load_word" then
|
||||
local dest = event_src(ev)
|
||||
if dest then
|
||||
local earlier = false
|
||||
for j = i - 1, 1, -1 do
|
||||
local prev = events[j]
|
||||
local prev_enc = prev.encoder or ""
|
||||
if prev_enc:match("^gte_cmdw_") then break end
|
||||
if prev_enc == "gte_mv_to_ctrl_r" then
|
||||
local prev_src = event_src(prev)
|
||||
local prev_alias = event_alias(prev)
|
||||
if prev_src == dest and prev_alias and prev_alias:match("^gte_cr_RT") then
|
||||
earlier = true
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
if earlier then
|
||||
for j = i + 1, #events do
|
||||
local later = events[j]
|
||||
local later_enc = later.encoder or ""
|
||||
if later_enc:match("^gte_cmdw_") then
|
||||
break
|
||||
end
|
||||
if later_enc == "gte_mv_to_ctrl_r" then
|
||||
local later_src = event_src(later)
|
||||
local later_alias = event_alias(later)
|
||||
if later_src == dest and later_alias and later_alias:match("^gte_cr_RT") then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name or "",
|
||||
line = ev.line or atom.line,
|
||||
check = "ctc2_chain_source_preservation",
|
||||
kind = "warning",
|
||||
msg = string.format(
|
||||
"atom '%s' reloads %s before a later ctc2 that still names it"
|
||||
, atom.name or "", dest),
|
||||
}
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return
|
||||
end
|
||||
local tokens = (atom.paths and atom.paths.tokens) or {}
|
||||
for i, t in ipairs(tokens) do
|
||||
local tok = t.tok or ""
|
||||
local ident = tok:match("^([%w_]+)") or ""
|
||||
if ident == "load_word" then
|
||||
local dest = tok:match("%(%s*([%w_]+)")
|
||||
if dest then
|
||||
for j = i + 1, #tokens do
|
||||
local later = tokens[j].tok or ""
|
||||
local later_ident = later:match("^([%w_]+)") or ""
|
||||
if later_ident:match("^gte_cmdw_") then
|
||||
break
|
||||
end
|
||||
if later_ident == "gte_mv_to_ctrl_r" then
|
||||
local later_src = later:match("%(%s*([%w_]+)")
|
||||
local later_alias = ctrl_alias_from_text(later)
|
||||
if later_src == dest and later_alias and later_alias:match("^gte_cr_RT") then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name or "",
|
||||
line = atom.line,
|
||||
check = "ctc2_chain_source_preservation",
|
||||
kind = "warning",
|
||||
msg = string.format(
|
||||
"atom '%s' reloads %s before a later ctc2 that still names it"
|
||||
, atom.name or "", dest),
|
||||
}
|
||||
break
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- check_immediate_field_width — flags integer literals passed to instruction
|
||||
-- macros that exceed the immediate field width. Reads `IMMEDIATE_FIELD_WIDTHS`
|
||||
-- from duffle.lua. Only fires on parseable integer literals; register names,
|
||||
@@ -2680,6 +3014,167 @@ local function check_immediate_field_width(atom, pipe_ctx, findings)
|
||||
end
|
||||
end
|
||||
|
||||
local SCRATCH_GPRS = {
|
||||
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
|
||||
R_AT = true, R_V0 = true, R_V1 = true,
|
||||
}
|
||||
|
||||
local function token_arg_list(tok)
|
||||
local inner = (tok or ""):match("%b()")
|
||||
if not inner then return {} end
|
||||
return duffle.split_top_level_commas(inner:sub(2, -2))
|
||||
end
|
||||
|
||||
local function arg_as_gpr(arg)
|
||||
arg = duffle.trim(arg or "")
|
||||
return arg:match("^R_[%w_]+$")
|
||||
end
|
||||
|
||||
local function collect_gpr_traffic(tokens)
|
||||
local reads, writes = {}, {}
|
||||
for _, t in ipairs(tokens or {}) do
|
||||
local tok = t.tok or t
|
||||
local ident = (tok or ""):match("^([%w_]+)") or ""
|
||||
if ident:sub(1, 4) ~= "mac_"
|
||||
and not (duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[ident])
|
||||
and ident ~= "nop" and ident ~= "atom_label" and ident ~= "atom_offset"
|
||||
then
|
||||
local args = token_arg_list(tok)
|
||||
local fx = (duffle.INSTRUCTION_GPR_EFFECTS or {})[ident]
|
||||
if fx then
|
||||
for _, pos in ipairs(fx.reads or {}) do
|
||||
local g = arg_as_gpr(args[pos])
|
||||
if g then reads[g] = true end
|
||||
end
|
||||
for _, pos in ipairs(fx.writes or {}) do
|
||||
local g = arg_as_gpr(args[pos])
|
||||
if g then writes[g] = true end
|
||||
end
|
||||
else
|
||||
for _, arg in ipairs(args) do
|
||||
local g = arg_as_gpr(arg)
|
||||
if g then
|
||||
reads[g] = true
|
||||
writes[g] = true
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return reads, writes
|
||||
end
|
||||
|
||||
local function gpr_set_from_list(list)
|
||||
local s = {}
|
||||
for _, name in ipairs(list or {}) do
|
||||
if type(name) == "string" then s[name] = true end
|
||||
end
|
||||
return s
|
||||
end
|
||||
|
||||
local function gpr_set_eq(a, b)
|
||||
for k in pairs(a) do if not b[k] then return false end end
|
||||
for k in pairs(b) do if not a[k] then return false end end
|
||||
return true
|
||||
end
|
||||
|
||||
local function gpr_set_keys(s)
|
||||
local keys = {}
|
||||
for k in pairs(s) do keys[#keys + 1] = k end
|
||||
table.sort(keys)
|
||||
return keys
|
||||
end
|
||||
|
||||
local function check_atom_calls_inferred_traffic(atom, pipe_ctx, findings)
|
||||
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||
if is_runtime_helper(atom) then return end
|
||||
local info = pipe_ctx.info_by_atom and pipe_ctx.info_by_atom[atom.name]
|
||||
if not info then return end
|
||||
if #(info.reads or {}) == 0 and #(info.writes or {}) == 0 then return end
|
||||
|
||||
local tokens = (atom.paths and atom.paths.tokens) or atom.body_tokens
|
||||
local reads, writes = collect_gpr_traffic(tokens)
|
||||
for _, t in ipairs(tokens or {}) do
|
||||
local ident = ((t.tok or t) or ""):match("^([%w_]+)") or ""
|
||||
if ident:sub(1, 4) == "mac_" then
|
||||
local bare = ident:sub(5)
|
||||
local idx = pipe_ctx.component_body_index and pipe_ctx.component_body_index[bare]
|
||||
local comp = (pipe_ctx.components_by_name or {})[bare]
|
||||
or (pipe_ctx.atoms_by_name or {})[bare]
|
||||
local body_toks = (idx and idx.body_tokens)
|
||||
or (comp and (comp.body_tokens or (comp.paths and comp.paths.tokens)))
|
||||
local cr, cw = collect_gpr_traffic(body_toks)
|
||||
for k in pairs(cr) do reads[k] = true end
|
||||
for k in pairs(cw) do writes[k] = true end
|
||||
end
|
||||
end
|
||||
|
||||
local decl_r = gpr_set_from_list(info.reads)
|
||||
local decl_w = gpr_set_from_list(info.writes)
|
||||
local function keep_inferred(inferred, declared)
|
||||
local out = {}
|
||||
for k in pairs(inferred) do
|
||||
if k == "R_0" then
|
||||
if declared[k] then out[k] = true end
|
||||
elseif SCRATCH_GPRS[k] then
|
||||
if declared[k] then out[k] = true end
|
||||
else
|
||||
out[k] = true
|
||||
end
|
||||
end
|
||||
return out
|
||||
end
|
||||
reads = keep_inferred(reads, decl_r)
|
||||
writes = keep_inferred(writes, decl_w)
|
||||
if not gpr_set_eq(decl_r, reads) or not gpr_set_eq(decl_w, writes) then
|
||||
findings[#findings + 1] = {
|
||||
atom = atom.name,
|
||||
line = info.info_line or atom.line,
|
||||
check = "atom_calls_inferred_traffic",
|
||||
kind = "warning",
|
||||
msg = string.format(
|
||||
"atom '%s' declared [%s]/[%s] != inferred [%s]/[%s]",
|
||||
atom.name,
|
||||
table.concat(gpr_set_keys(decl_r), ","),
|
||||
table.concat(gpr_set_keys(decl_w), ","),
|
||||
table.concat(gpr_set_keys(reads), ","),
|
||||
table.concat(gpr_set_keys(writes), ",")),
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
local function check_component_self_consistency(src, pipe_ctx, findings)
|
||||
local first = pipe_ctx.source_order and pipe_ctx.source_order[1]
|
||||
if first and src ~= first then return end
|
||||
local infos = pipe_ctx.component_atom_infos or {}
|
||||
local atoms_by_name = pipe_ctx.atoms_by_name or {}
|
||||
for _, ai in ipairs(infos) do
|
||||
local name = ai.atom_name or ai.name
|
||||
local atom = name and atoms_by_name[name]
|
||||
if atom and not atom.debug_skip then
|
||||
local tokens = (atom.paths and atom.paths.tokens) or atom.body_tokens
|
||||
local reads, writes = collect_gpr_traffic(tokens)
|
||||
local decl_r = gpr_set_from_list(ai.reads)
|
||||
local decl_w = gpr_set_from_list(ai.writes)
|
||||
if not gpr_set_eq(decl_r, reads) or not gpr_set_eq(decl_w, writes) then
|
||||
findings[#findings + 1] = {
|
||||
atom = name,
|
||||
line = ai.info_line or atom.line,
|
||||
check = "component_self_consistency",
|
||||
kind = "warning",
|
||||
msg = string.format(
|
||||
"component '%s' atom_reads/atom_writes [%s]/[%s] != body [%s]/[%s]",
|
||||
name,
|
||||
table.concat(gpr_set_keys(decl_r), ","),
|
||||
table.concat(gpr_set_keys(decl_w), ","),
|
||||
table.concat(gpr_set_keys(reads), ","),
|
||||
table.concat(gpr_set_keys(writes), ",")),
|
||||
}
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
@@ -2707,12 +3202,17 @@ local CHECK_RULES = {
|
||||
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
|
||||
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
|
||||
{ name = "gte_cr_alias_writes", per_atom = check_gte_cr_alias_writes },
|
||||
{ name = "gte_cr_alias_writes_xatom", per_source = check_gte_cr_alias_writes_xatom },
|
||||
{ name = "gte_packed_writes", per_atom = check_gte_packed_writes },
|
||||
{ name = "ctc2_chain_source_preservation", per_atom = check_ctc2_chain_source_preservation },
|
||||
{ name = "rtdiagonal_completeness", per_atom = check_rtdiagonal_completeness },
|
||||
{ name = "gte_cr_TR_naming", per_atom = check_gte_cr_TR_naming },
|
||||
{ name = "immediate_field_width", per_atom = check_immediate_field_width },
|
||||
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
|
||||
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
|
||||
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
|
||||
{ name = "component_self_consistency", per_source = check_component_self_consistency },
|
||||
{ name = "atom_calls_inferred_traffic", per_atom = check_atom_calls_inferred_traffic },
|
||||
}
|
||||
|
||||
-- ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -2749,6 +3249,11 @@ local function build_corpus_pipe_ctx(ctx)
|
||||
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
|
||||
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
|
||||
components_by_name = corpus.components or {},
|
||||
atoms_by_name = corpus.atoms_by_name or {},
|
||||
tape_chains = corpus.tape_chains or {},
|
||||
source_order = corpus.source_order or {},
|
||||
component_atom_infos = corpus.component_atom_infos or {},
|
||||
atom_infos = corpus.atom_infos or {},
|
||||
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
||||
atom_infos_list = corpus.atom_infos or {},
|
||||
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
||||
@@ -2807,6 +3312,11 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
||||
type_name_registry = corpus_pipe_ctx.type_name_registry,
|
||||
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
|
||||
components_by_name = corpus_pipe_ctx.components_by_name,
|
||||
atoms_by_name = corpus_pipe_ctx.atoms_by_name,
|
||||
tape_chains = corpus_pipe_ctx.tape_chains,
|
||||
source_order = corpus_pipe_ctx.source_order,
|
||||
component_atom_infos = corpus_pipe_ctx.component_atom_infos,
|
||||
atom_infos_all = corpus_pipe_ctx.atom_infos,
|
||||
}
|
||||
-- Shared cross-source component-body index is owned by the corpus (`corpus.component_body_index`, populated by `passes/components.lua`).
|
||||
-- Per-atom checks consume the corpus-owned index directly.
|
||||
|
||||
Reference in New Issue
Block a user