32 changed files with 2768 additions and 1072 deletions
+43
View File
@@ -0,0 +1,43 @@
# Package and install the local VS Code Insiders extensions under .vscode/.
# Usage:
# .\install_extensions.ps1
# .\install_extensions.ps1 -SkipPackage
param([switch] $SkipPackage)
$path_vscode = $PSScriptRoot
$code_insiders = "C:\apps\Microsoft VS Code Insiders\bin\code-insiders.cmd"
if (-not (test-path -literalpath $code_insiders)) {
$found = get-command code-insiders -erroraction silentlycontinue
if ($found) { $code_insiders = $found.source }
}
if (-not (test-path -literalpath $code_insiders)) { throw "code-insiders not found. Install VS Code Insiders or add it to PATH." }
$extensions = @(
(join-path $path_vscode "tape-atom-syntax"),
(join-path $path_vscode "cozy-and-windy")
)
foreach ($extension in $extensions) {
$package_json = join-path $extension "package.json"
if (-not (test-path -literalpath $package_json)) { throw "missing $package_json" }
$manifest = get-content -literalpath $package_json -raw | convertfrom-json
$vsix = join-path $extension ("{0}-{1}.vsix" -f $manifest.name, $manifest.version)
if (-not $SkipPackage) {
if (-not $manifest.scripts.package) { throw "$package_json has no scripts.package" }
write-host "packaging $($manifest.displayName) ($($manifest.name)@$($manifest.version))"
& npm --prefix $extension run package
if ($LASTEXITCODE -ne 0) { throw "npm run package failed for $extension" }
}
if (-not (test-path -literalpath $vsix)) { throw "missing $vsix" }
write-host "installing $vsix"
& $code_insiders --install-extension $vsix --force
if ($LASTEXITCODE -ne 0) { throw "install failed for $vsix" }
}
write-host "done. reload the Insiders window (Developer: Reload Window)."
Binary file not shown.
+1 -1
View File
@@ -49,7 +49,7 @@ const DSL_KEYWORDS = new Set([
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
]);
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_"]);
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
@@ -56,7 +56,7 @@
"name": "support.function.duffle.annotation"
},
"delay-slots": {
"match": "\\b(LdSlot_|BdSlot_)\\b",
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
"name": "keyword.operator.duffle.delayslot"
},
"types": {
+10
View File
@@ -105,6 +105,16 @@ test("document-local declarations override an empty workspace index", () => {
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
});
test("delay slot markers share the tapeDelaySlot token", () => {
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
});
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
const source = "atom_reads(R_A /* broken";
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
+10 -9
View File
@@ -140,16 +140,17 @@ enum { false = 0, true = 1, true_overflow, };
typedef void Proc_(VoidFn) (void);
#define Kilo_(n) (C_(U4, n) << 10)
#define Mega_(n) (C_(U4, n) << 20)
#define Giga_(n) (C_(U4, n) << 30)
#define Tera_(n) (C_(U4, n) << 40)
#define Kilo_(n) (C_(U4, n) << 10)
#define Mega_(n) (C_(U4, n) << 20)
#define Giga_(n) (C_(U4, n) << 30)
#define Tera_(n) (C_(U4, n) << 40)
#define null C_(U4, 0)
#define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OT_(field) O_(typeof_ptr(& field), field))
#define S_(data) C_(U4, sizeof(data))
#define null C_(U4, 0)
#define nullptr C_(void*, 0)
#define O_(type, field) C_(U4, & C_(type*,0)->field)
#define OA_(type, aexpr) C_(U4, & C_(type*,0) aexpr)
#define OT_(field) O_(typeof_ptr(& field), field))
#define S_(data) C_(U4, sizeof(data))
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
#define sop_2(op,a,b) C_(U2, s2_(a) op s2_(b))
+102 -36
View File
@@ -17,7 +17,7 @@
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
@@ -58,10 +58,21 @@ WORD_COUNT(mac_yield_load, 1)
, nop
WORD_COUNT(mac_yield_tail, 3)
/* atom_dbg_skip */
#define mac_load_half_v3(tx, ty, tz, base, offset) \
load_half(tx, base, offset + OA_(U2,[0])) \
, load_half(ty, base, offset + OA_(U2,[1])) \
, load_half(tz, base, offset + OA_(U2,[2]))
WORD_COUNT(mac_load_half_v3, 3)
#define mac_load_v3s2(transfer, base, offset) \
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_load_v3s2, 3)
/* atom_dbg_skip */
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
load_half( rs_x, r_base, offset + O_(V3_S2,x)) \
, load_half( rs_y, r_base, offset + O_(V3_S2,y))
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
, load_half(rs_y, r_base, offset + O_(V3_S2,y))
WORD_COUNT(mac_load_v2s2, 2)
/* atom_dbg_skip */
@@ -71,26 +82,75 @@ WORD_COUNT(mac_load_v2s2, 2)
WORD_COUNT(mac_store_v2s2, 2)
/* atom_dbg_skip */
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \
, load_word( rs_z, r_base, offset + O_(V3_S4,z))
#define mac_load_word_v3(tx, ty, tz, base, offset) \
load_word(tx, base, offset + OA_(U4,[0])) \
, load_word(ty, base, offset + OA_(U4,[1])) \
, load_word(tz, base, offset + OA_(U4,[2]))
WORD_COUNT(mac_load_word_v3, 3)
#define mac_load_v3s4(transfer, base, offset) \
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_load_v3s4, 3)
/* atom_dbg_skip */
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
store_word(rt_x, base, offset + O_(V3_S4,x)) \
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
, store_word(rt_z, base, offset + O_(V3_S4,z))
WORD_COUNT(mac_store_v3s4, 3)
#define mac_load_p3s4(transfer, base, offset) \
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_load_p3s4, 3)
/* atom_dbg_skip */
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
sub_s(rds_x, rds_x, rt_x) \
, sub_s(rds_y, rds_y, rt_y) \
, sub_s(rds_z, rds_z, rt_z)
#define mac_store_half_v3(tx, ty, tz, base, offset) \
store_half(tx, base, offset + OA_(U2,[0])) \
, store_half(ty, base, offset + OA_(U2,[1])) \
, store_half(tz, base, offset + OA_(U2,[2]))
WORD_COUNT(mac_store_half_v3, 3)
#define mac_store_v3s2(transfer, base, offset) \
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_store_v3s2, 3)
/* atom_dbg_skip */
#define mac_store_word_v3(tx, ty, tz, base, offset) \
store_word(tx, base, offset + OA_(U4,[0])) \
, store_word(ty, base, offset + OA_(U4,[1])) \
, store_word(tz, base, offset + OA_(U4,[2]))
WORD_COUNT(mac_store_word_v3, 3)
#define mac_store_v3s4(transfer, base, offset) \
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_store_v3s4, 3)
#define mac_store_p3s4(transfer, base, offset) \
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
WORD_COUNT(mac_store_p3s4, 3)
/* atom_dbg_skip */
#define mac_add_si_v3s4(rt_x, rt_y, rt_z, base, offset) \
add_si(rt_x, base, O_(V3_S4,x)) \
, add_si(rt_y, base, O_(V3_S4,y)) \
, add_si(rt_z, base, O_(V3_S4,z))
WORD_COUNT(mac_add_si_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
sub_s(dx, sx, tx) \
, sub_s(dy, sy, ty) \
, sub_s(dz, sz, tz)
WORD_COUNT(mac_sub_s_v3, 3)
#define mac_sub_v3s4(d, s, t) \
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
WORD_COUNT(mac_sub_v3s4, 3)
/* atom_dbg_skip */
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
sub_s(ds_x, ds_x, tx) \
, sub_s(ds_y, ds_y, ty) \
, sub_s(ds_z, ds_z, tz)
WORD_COUNT(mac_sub_s_v3_self, 3)
#define mac_sub_v3s4_self(ds, t) \
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
WORD_COUNT(mac_sub_v3s4_self, 3)
/* atom_dbg_skip */
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
@@ -105,6 +165,24 @@ WORD_COUNT(mac_store_rects2, 4)
, or_i_self( dst, u4_lo(imm))
WORD_COUNT(mac_load_word_imm, 2)
#define mac_shift_aright_v3_self(dt_x, dt_y, dt_z, shift_amount) \
shift_aright(dt_x, dt_x, shift_amount) \
, shift_aright(dt_y, dt_y, shift_amount) \
, shift_aright(dt_z, dt_z, shift_amount)
WORD_COUNT(mac_shift_aright_v3_self, 3)
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
shift_aright_var(rd_v0, rs_v0, r_shift) \
, shift_aright_var(rd_v1, rs_v1, r_shift) \
, shift_aright_var(rd_v2, rs_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3, 3)
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
shift_aright_var(rds_v0, rds_v0, r_shift) \
, shift_aright_var(rds_v1, rds_v1, r_shift) \
, shift_aright_var(rds_v2, rds_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
/* atom_dbg_skip */
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
@@ -125,19 +203,19 @@ WORD_COUNT(mac_gte_store_f3, 3)
, add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY0) \
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
, gte_mv_to_data_r(R_V1, C2_VZ0) \
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY1) \
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
, gte_mv_to_data_r(R_V1, C2_VZ1) \
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
, add_u_self(R_AT, r_vert_base) \
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
, gte_mv_to_data_r(R_V0, C2_VXY2) \
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
, gte_mv_to_data_r(R_V1, C2_VZ2)
WORD_COUNT(mac_gte_load_tri_verts, 18)
@@ -162,11 +240,11 @@ WORD_COUNT(mac_gte_store_g4_p3, 1)
WORD_COUNT(mac_gte_sqr_v3, 8)
/* atom_dbg_skip */
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop_slot) \
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
gte_mv_to_data_r(r_sx, C2_IR1) \
, gte_mv_to_data_r(r_sy, C2_IR2) \
, gte_mv_to_data_r(r_sz, C2_IR3) \
, nop_slot \
, delay_slot \
, gte_cmdw_sqr
WORD_COUNT(mac_gte_sqr_v3s4, 5)
@@ -204,25 +282,13 @@ WORD_COUNT(mac_trans_mt3s3s4, 6)
, shift_aright(r_mag_sq, r_mag_sq, 1)
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
shift_aright_var(rd_v0, rs_v0, r_shift) \
, shift_aright_var(rd_v1, rs_v1, r_shift) \
, shift_aright_var(rd_v2, rs_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3, 3)
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
shift_aright_var(rds_v0, rds_v0, r_shift) \
, shift_aright_var(rds_v1, rds_v1, r_shift) \
, shift_aright_var(rds_v2, rds_v2, r_shift)
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
gte_mv_to_data_r(to_ir0, C2_IR0) \
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
, gte_mv_to_data_r(to_ir2, C2_IR2) \
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
, LdSlot_ nop_slot1 \
, LdSlot_ nop_slot2 \
, GteDelay_ nop_slot1 \
, GteDelay_ nop_slot2 \
, gte_cmdw_gpf \
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
+10 -2
View File
@@ -14,7 +14,7 @@
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
@@ -25,7 +25,15 @@
#pragma region duffle
// --- atom: normalize_v3s4 (47 words) ---
// --- atom: example_atom_proc (10 words) ---
#define _atom_offset_example_atom_proc_skip 2
enum {
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
};
// --- atom: build_normalize_v3s4 (61 words) ---
#define _atom_offset_aligned_done_srav_path 3
#define _atom_offset_srav_path_aligned_done 4
+2 -1
View File
@@ -44,7 +44,8 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
})
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ab, {
// TODO(Ed): Expose R_T1 as a r_t0, r_V0 as r_t2
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_cursor, U2 poly_size) MipsAtomComp_Proc_(ab, {
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
+4 -6
View File
@@ -68,6 +68,7 @@ enum {
#define gp0_send(word) (HW_GP0[0] = (word))
#define gp1_send(word) (HW_GP1[0] = (word))
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
/* ============================================================================
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
@@ -418,14 +419,11 @@ typedef Struct_(PolyTag) {
};
};
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier is `R_` (restrict) or `V_` (volatile).
* No raw C-style casts. RHS values are assumed to be `U4` — caller passes a `U4` directly. */
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
/* `set_code` is no longer in the new PolyTag design — the code byte lives in the primitive body
/* `set_code` is no longer in the new PolyTag design
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters,
* which set both the tag's length and the code. */
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, which set both the tag's length and the code. */
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
@@ -572,7 +570,7 @@ enum {
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
gp0_tpage_default = 10,
/* TPage semi-transparency mode payload values (NOT bit positions). */
/* TPage semi-transparency mode payload values. */
gp0_tpage_semi_trans_none = 0x0,
gp0_tpage_semi_trans_alpha = 0x1,
gp0_tpage_semi_trans_add = 0x2,
+30 -44
View File
@@ -28,9 +28,9 @@ FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
})
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
@@ -38,7 +38,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
@@ -65,12 +65,12 @@ FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4
* The SQR command always squares IR1/IR2/IR3 — those C2 registers are fixed.
* The GPRs holding the source vector are caller-determined.
* Words: 5 (3 mtc2 + 1 nop hazard + 1 cmd). */
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode nop_slot)
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(r_sx, C2_IR1),
gte_mv_to_data_r(r_sy, C2_IR2),
gte_mv_to_data_r(r_sz, C2_IR3),
nop_slot, gte_cmdw_sqr,
delay_slot, gte_cmdw_sqr,
})
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
@@ -132,8 +132,7 @@ FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
U4 r_shift,
U4 r_mag_sq,
U4 r_mag_sq_copy
)
U4 r_mag_sq_copy)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
and_i(r_shift, r_shift, gte_lzcr_even_mask),
or_u(r_mag_sq_copy, r_mag_sq, 0),
@@ -142,25 +141,6 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
shift_aright(r_mag_sq, r_mag_sq, 1),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
, Reg rd_v0, Reg rd_v1, Reg rd_v2
, Reg rs_v0, Reg rs_v1, Reg rs_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rd_v0, rs_v0, r_shift),
shift_aright_var(rd_v1, rs_v1, r_shift),
shift_aright_var(rd_v2, rs_v2, r_shift),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab
, Reg rds_v0, Reg rds_v1, Reg rds_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rds_v0, rds_v0, r_shift),
shift_aright_var(rds_v1, rds_v1, r_shift),
shift_aright_var(rds_v2, rds_v2, r_shift),
})
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
@@ -170,17 +150,16 @@ MipsAtomComp_Proc_(ab, {
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
gte_mv_to_data_r(to_ir2, C2_IR2),
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
LdSlot_ nop_slot1,
LdSlot_ nop_slot2,
GteDelay_ nop_slot1,
GteDelay_ nop_slot2,
gte_cmdw_gpf,
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
gte_mv_from_data_r(fr_mac3, C2_MAC3),
})
FI_ Slice_MipsCode gte_mv_from_data_r_mac123(AtomBuilder_R ab
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
)
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
MipsAtomComp_Proc_(ab, {
gte_mv_from_data_r(fr_mac1, C2_MAC1),
gte_mv_from_data_r(fr_mac2, C2_MAC2),
@@ -257,9 +236,13 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
};
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
typedef Struct_(RegUse_normalize_v3s4_proc) {
Reg scratch; // Scratch base carrier.
typedef Struct_(Binds_build_normalize_v3s4) {
U4 scratch;
U2 src_offset;
U2 dst_offset;
};
typedef Struct_(RegUse_build_normalize_v3s4) {
Reg scratch;
Reg src_ptr;
Reg dst_ptr;
Reg recip_est; // |v|² sum + shift-input + sqrtbl[index]
@@ -267,7 +250,7 @@ typedef Struct_(RegUse_normalize_v3s4_proc) {
Reg src_x;
union { Reg mac1_scratch; } t3;
union { Reg mac2_scratch; } t4;
union { Reg shift_count, btarget, lookup_addr, src_z; } t5;
union { Reg btarget, shift_count, lookup_addr, src_z; } t5;
};
/* ─── Full normalize (all 4 stages inline) ───
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav).
@@ -300,13 +283,16 @@ typedef Struct_(RegUse_normalize_v3s4_proc) {
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
*/
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_normalize_v3s4_proc r)
internal MipsAtom* build_normalize_v3s4(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_build_normalize_v3s4 r)
MipsAtom_Proc_(aa, {
add_si(r.src_ptr, r.scratch, src_offset), /* r_src_ptr = &src */
// load_word(r.scratch, R_TapePtr, O_(Binds_build_normalize_v3s4,scratch)),
// add_ui_self(R_TapePtr, S_(Binds_build_normalize_v3s4)),
add_si(r.src_ptr, r.scratch, src_offset), /* r_src_ptr = &src */
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
* r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */
mac_load_v3s4(r.src_x, r.recip_est, r.t5.lookup_addr, r.src_ptr, 0),
mac_load_word_v3(r.src_x, r.recip_est, r.t5.src_z, r.src_ptr, 0),
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
@@ -315,8 +301,8 @@ MipsAtom_Proc_(aa, {
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
add_u_self( r.norm, r.t3.mac1_scratch),
add_u_self( r.norm, r.t4.mac2_scratch),
gte_mv_to_data_r( r.norm, C2_LZCS), LdSlot_ nop2,
gte_mv_from_data_r(r.shift, C2_LZCR), LdSlot_ nop,
gte_mv_to_data_r( r.norm, C2_LZCS), GteDelay_ nop2,
gte_mv_from_data_r(r.shift, C2_LZCR), GteDelay_ nop,
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
@@ -350,13 +336,13 @@ MipsAtom_Proc_(aa, {
r.recip_est,
r.t5.src_z, /* IR3 = src.z (reloaded) */
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
LdSlot_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
LdSlot_ nop
GteDelay_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
GteDelay_ nop
),
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
mac_store_v3s4(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
mac_store_word_v3(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
mac_yield()
})
+24 -30
View File
@@ -33,7 +33,7 @@
* gte.h — Geometry Transformation Engine (COP2) for the PS1
* ============================================================================
*
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C.
* Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
* No GCC inline-assembly string syntax in the code body.
*
* STYLE NOTES
@@ -191,27 +191,22 @@ enum {
};
/* --- GTE Control Register Aliases (Pitfall 1) ---
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon:
* Three pairs of aliases map to the C2 control-register slot:
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
* Cross-alias writes inside one atom body, or across the wave-context boundary,
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes
* (CHECK_RULES row) warns about each pair per source. See
* docs/gte_reference.md §"Control-register alias table" for the silicon
* rationale and the libgte outer-product convention.
* Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
* The metaprogram's check_gte_cr_alias_writes (CHECK_RULES row) warns about each pair per source.
* See psx-spx docs/gte_reference.md §"Control-register alias table" for the silicon rationale and the libgte outer-product convention.
*/
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
* for the canonical write pattern.
* OP and MVMVA read D1/D2/D3 from these packed slots.
* The libgte outer-product convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then C2[4] in sequence;
* the SECOND write's low half is RT22, not RT13.
*/
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
@@ -300,8 +295,7 @@ enum { _C2_TX_SUBS_ = 0
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
/* GTE Data vs Control Register Transfers
*
* Each macro emits a single .word constant for one of MFC2/CFC2/MTC2/CTC2.
* Each macro emits a single instruction for one of MFC2/CFC2/MTC2/CTC2.
*
* `rd` is the C2 register index in the file the sub-opcode names:
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
@@ -316,14 +310,14 @@ enum { _C2_TX_SUBS_ = 0
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
* - rs: GPR base address
* - rt: COP2 data register index (0..31)
* - imm: signed 16-bit offset
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked
* into a .word — use the string-style `gte_load_v0` macro below instead. */
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked into a .word — use the string-style `gte_load_v0` macro below instead. */
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
/* Store Word */
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
@@ -332,8 +326,7 @@ enum { _C2_TX_SUBS_ = 0
* `swc2` is redundant when we're already inside the `gte_` namespace.
* gte_lw rt, base, off → lwc2 rt, off(base)
* gte_sw rt, base, off → swc2 rt, off(base)
* For the typical user-facing vector-level load (xy + z as two instructions),
* use the higher-level `gte_load_vN` macros below. */
* For the typical user-facing vector-level load (xy + z as two instructions), use the higher-level `gte_load_vN` macros below. */
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
@@ -350,13 +343,13 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
#define enc_gte_sf(sf) ((sf) << gte_shift_sf )
#define enc_gte_mx(mx) ((mx) << gte_shift_mx )
#define enc_gte_v(v) ((v) << gte_shift_v )
#define enc_gte_cv(cv) ((cv) << gte_shift_cv )
#define enc_gte_lm(lm) ((lm) << gte_shift_lm )
#define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
#define enc_gte_sf(sf) ((sf) << gte_shift_sf )
#define enc_gte_mx(mx) ((mx) << gte_shift_mx )
#define enc_gte_v(v) ((v) << gte_shift_v )
#define enc_gte_cv(cv) ((cv) << gte_shift_cv )
#define enc_gte_lm(lm) ((lm) << gte_shift_lm )
#define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
#define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
/* Composite: all six GTE fields + the COP2/CO base. */
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
@@ -408,10 +401,11 @@ enum { _C2_TX_SUBS_ = 0
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- PSY-Q terminology */
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. */
#define gte_cmdw_cross gte_cmdw_op /* "cross product" -- geometric-algebra terminology.
* RGA(Lengyel): The GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
* The wedge alias is a 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
+27 -2
View File
@@ -101,11 +101,12 @@ enum {
};
typedef U2 Reg; // Register parameter used with atom or atom component procedures
#define Reg_(type) tmpl(Reg,type) // Just a way to template register allocations of C-struct types.
typedef U4 const MipsCode; // Underlying type to mips asm words.
typedef Slice_(MipsCode);
typedef U4 const MipsAtom;
typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
typedef Slice_(MipsAtom);
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
// MipsAtom* <identifier>[...];
@@ -143,6 +144,8 @@ typedef Slice_(MipsAtom);
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
Files containing only atoms and atom components.
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
@@ -242,6 +245,7 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
add_ui_self(R_TapePtr, S_(MipsCode)),
jump_reg( R_AtomJmp), nop,
};
#pragma endregion Macro Atom Components
#pragma region Atom Builder
@@ -371,13 +375,34 @@ FI_ void regfile_reset(RegFile_R rf) {
rf->GPR[0] = u4_lo(regfile_abi_mask);
rf->GPR[1] = u4_hi(regfile_abi_mask);
}
FI_ void regfile_reset_mask(RegFile_R rf, U4 mask) {
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
rf->GPR[0] = u4_lo(mask);
rf->GPR[1] = u4_hi(mask);
}
#pragma endregion RegFileArena (Register File Allocator)
#pragma region Mips Atom Procs
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
Unlike the usual enum-based declarations, they provide a namespaced scope
and have view types via union declarations.
*/
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
typedef Struct_(RegUse_example_atom_proc) {
Reg const ro_register; // Scratch base carrier.
Reg usual_modifiable;
union { Reg view_1, view_2, view_3; } t1;
};
internal MipsAtom* example_atom_proc(AtomArena_R aa, U2 offset, RegUse_example_atom_proc r)
MipsAtom_Proc_(aa, {
add_si(r.usual_modifiable, r.ro_register, offset),
or_u(r.t1.view_1, r.ro_register, 0),
branch_lt_zero(r.t1.view_1, atom_offset(example_atom_proc, skip)), BdSlot_ nop,
li_s(r.t1.view_2, 100),
atom_label(skip)
add_si(r.t1.view_3, r.usual_modifiable, 10),
mac_yield(),
})
#pragma endregion Mips Atom Procs
-55
View File
@@ -1,55 +0,0 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "math.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#pragma region MACs (Mips Atom Component)
// FI_ Slice_MipsCode ac_load_imm
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_load_p3s4 ac_load_v3s4
#define mac_load_p3s4 mac_load_v3s4
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_word(rt_x, base, offset + O_(V3_S4,x)),
store_word(rt_y, base, offset + O_(V3_S4,y)),
store_word(rt_z, base, offset + O_(V3_S4,z)),
})
// TODO(Ed): we could generate these mappings properly..
#define ac_store_p3s4 ac_store_v3s4
#define mac_store_p3s4 mac_store_v3s4
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(rds_x, rds_x, rt_x),
sub_s(rds_y, rds_y, rt_y),
sub_s(rds_z, rds_z, rt_z),
})
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
store_half(rt_height, base, offset + O_(Rect_S2,height)),
})
#pragma endregion MACs (Mips Atom Component)
+96
View File
@@ -0,0 +1,96 @@
#ifdef INTELLISENSE_DIRECTIVES
# include "gen/macs.h"
# include "gen/offsets.h"
# include "math.h"
# include "lottes_tape.h"
#endif
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
#pragma region MACs (Mips Atom Component)
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half(tx, base, offset + OA_(U2,[0])),
load_half(ty, base, offset + OA_(U2,[1])),
load_half(tz, base, offset + OA_(U2,[2])),
})
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
})
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(V2_S2,x)),
store_half(rt_y, base, offset + O_(V2_S2,y)),
})
FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
load_word(tx, base, offset + OA_(U4,[0])),
load_word(ty, base, offset + OA_(U4,[1])),
load_word(tz, base, offset + OA_(U4,[2])),
})
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(tx, base, offset + OA_(U2,[0])),
store_half(ty, base, offset + OA_(U2,[1])),
store_half(tz, base, offset + OA_(U2,[2])),
})
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_word(tx, base, offset + OA_(U4,[0])),
store_word(ty, base, offset + OA_(U4,[1])),
store_word(tz, base, offset + OA_(U4,[2])),
})
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_store_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
FI_ Slice_MipsCode ac_add_si_v3s4(AtomBuilder_R ab, Reg rt_x, Reg rt_y, Reg rt_z, Reg base, U2 offset)
atom_dbg_skip MipsAtomComp_Proc_(ab, {
add_si(rt_x, base, O_(V3_S4,x)),
add_si(rt_y, base, O_(V3_S4,y)),
add_si(rt_z, base, O_(V3_S4,z)),
})
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
, Reg dx, Reg dy, Reg dz
, Reg sx, Reg sy, Reg sz
, Reg tx, Reg ty, Reg tz
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(dx, sx, tx),
sub_s(dy, sy, ty),
sub_s(dz, sz, tz),
})
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
sub_s(ds_x, ds_x, tx),
sub_s(ds_y, ds_y, ty),
sub_s(ds_z, ds_z, tz),
})
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
store_half(rt_x, base, offset + O_(Rect_S2,x)),
store_half(rt_y, base, offset + O_(Rect_S2,y)),
store_half(rt_width, base, offset + O_(Rect_S2,width)),
store_half(rt_height, base, offset + O_(Rect_S2,height)),
})
#pragma endregion MACs (Mips Atom Component)
+26
View File
@@ -16,6 +16,32 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
or_i_self( dst, u4_lo(imm)),
})
FI_ Slice_MipsCode ac_shift_aright_v3_self(AtomBuilder_R ab, Reg dt_x, Reg dt_y, Reg dt_z, U2 shift_amount)
MipsAtomComp_Proc_( ab, {
shift_aright(dt_x, dt_x, shift_amount),
shift_aright(dt_y, dt_y, shift_amount),
shift_aright(dt_z, dt_z, shift_amount),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
, Reg rd_v0, Reg rd_v1, Reg rd_v2
, Reg rs_v0, Reg rs_v1, Reg rs_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rd_v0, rs_v0, r_shift),
shift_aright_var(rd_v1, rs_v1, r_shift),
shift_aright_var(rd_v2, rs_v2, r_shift),
})
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab
, Reg rds_v0, Reg rds_v1, Reg rds_v2
, Reg r_shift)
MipsAtomComp_Proc_(ab, {
shift_aright_var(rds_v0, rds_v0, r_shift),
shift_aright_var(rds_v1, rds_v1, r_shift),
shift_aright_var(rds_v2, rds_v2, r_shift),
})
#pragma endregion MACs (Mips Atom Components)
#pragma region Baked Atoms
+2 -2
View File
@@ -64,8 +64,8 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
asm_words(
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
call_reg(R_T2), /* jalr $t2, $ra */
nop /* BD slot */
call_reg(R_T2), /* jalr $t2, $ra */
nop /* BD slot */
)
asm_clobber:
rlit(R_AT),
+2 -2
View File
@@ -8,7 +8,7 @@
#pragma region hello_camera
// --- atom: pad_input_cube_rotation (60 words) ---
// --- atom: pad_input_cube_rotation (61 words) ---
#define _atom_offset_dpad_left_exit_dpad_left 6
#define _atom_offset_dpad_right_exit_dpad_right 6
@@ -44,7 +44,7 @@ enum {
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
};
// --- atom: cube_g4_face (76 words) ---
// --- atom: cube_g4_face (75 words) ---
#define _atom_offset_cull_cube_g4_face_exit 41
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
+219 -308
View File
@@ -10,7 +10,7 @@
# include "duffle/pad.h"
# include "duffle/word_count.metadata.h"
# include "duffle/psyq.h"
# include "duffle/math.atom.c"
# include "duffle/math.atom.h"
# include "duffle/mips.atom.c"
# include "duffle/gte.atom.c"
# include "duffle/gp.atom.c"
@@ -94,6 +94,26 @@ MipsAtomComp_Proc_(ab, {
#pragma region Atom Procs
// Modular Atoms
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
#pragma region resolve_look_at
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
typedef AtomBundle_(resolve_look_at) { MipsAtom*
input_and_sub,
normalize_fwd_uz,
cross_uz_up_into_right,
normalize_right_ux,
cross_uz_ux_to_up,
normalize_up_uy,
populate,
set_gte_mt3s2s4,
matrix_vector,
trans_matrix;
};
enum {
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
@@ -119,15 +139,17 @@ typedef Struct_(ResolveLookAtScratch) {
V3_S4 up_in; /* offset +128 (16 bytes) */
};
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
typedef Struct_(Binds_ResolveLookAtSub) {
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
ResolveLookAtScratch* scratchpad;
};
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
Reg scratch;
Reg target; Reg eye; Reg up_in;
Reg t0; Reg t1; Reg t2; Reg t3; Reg t4;
};
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
* Staging work:
* * Stage eye.x/y/z → scratch (for atom 6's translation column)
@@ -146,66 +168,63 @@ typedef Struct_(Binds_ResolveLookAtSub) {
* R_V0 : hardcoded (load eye.z / target.z)
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
*/
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa,
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
U4 r_scratch
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
) MipsAtom_Proc_(aa, {
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_eye_ptr, 0),
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye)),
// internal MipsAtom* resolve_look_at_input_and_sub(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
load_word(r.target, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
load_word(r.eye, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
load_word(r.up_in, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
load_word(r.scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
/* Stage up_in.x/y/z into the scratchpad. */
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_up_in_ptr, 0),
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in)),
mac_load_word_v3( r.t0, r.t1, r.t2, r.up_in, 0), LdSlot_
mac_store_word_v3(r.t0, r.t1, r.t2, r.scratch, O_(ResolveLookAtScratch,up_in)),
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
mac_load_word_v3( r.t0, r.t1, r.t2, r.eye, 0), LdSlot_
mac_store_word_v3(r.t0, r.t1, r.t2, r.scratch, O_(ResolveLookAtScratch,eye)),
/* Compute fwd = target - eye. */
mac_load_p3s4(r_tmp0, r_tmp1, r_tmp2, r_target_ptr, 0),
mac_load_p3s4(r_tmp3, R_AT, R_V0, r_eye_ptr, 0),
mac_sub_v3s4(
r_tmp0, r_tmp1, r_tmp2,
r_tmp3, R_AT, R_V0),
mac_store_v3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd)),
// mac_load_p3s4(t3, R_AT, t4, r.eye, 0),
mac_load_word_v3(r.t3, R_AT, r.t4, r.target, 0), LdSlot_
mac_sub_s_v3_self(
r.t3, R_AT, r.t4,
r.t0, r.t1, r.t2),
mac_store_word_v3(r.t3, R_AT, r.t4, r.scratch, O_(ResolveLookAtScratch,fwd)),
mac_yield()
})
typedef Struct_(RegUse_resolve_look_at_cross_uz_up_into_right) {
Reg scratch;
Reg a; Reg b; Reg c; /* load a.x/y/z; result out.x/y/z */
Reg d; /* load b.x */
Reg f; /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
union { Reg t1, g, target0; };
union { Reg t2, h, target1; };
Reg t0;
};
/* Atom 2: cross uz × up_in → right. */
internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
// internal MipsAtom* AtomBundleEntry_(resolve_look_at, cross_uz_up_to_right)(AtomArena_R aa,
internal MipsAtom* resolve_look_at_cross_uz_up_into_right(AtomArena_R aa,
RegUse_resolve_look_at_cross_uz_up_into_right r
) MipsAtom_Proc_(aa, {
/* FIX: build packed RT22+RT33 with proper sign extension. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
add_si(r.g, r.scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r.h, r.scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
add_si(r.f, r.scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
nop,
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
mac_load_word_v3(r.a, r.b, r.c, r.g, 0), LdSlot_
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
/* Save the two RT control-register slots OP will clobber. We reuse
* r_g/r_h (scratch pointers, no longer needed) as the save targets. */
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
mac_load_word_v3(r.d, R_AT, r.t0, r.h, 0), LdSlot_ // (taken by gte_mv_from_ctrl_r)
/* Save the two RT control-register slots OP will clobber. We reuse r_g/r_h (scratch pointers, no longer needed) as the save targets. */
gte_mv_from_ctrl_r(r.target0, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
gte_mv_from_ctrl_r(r.target1, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
/* Load uz.x/uz.y/uz.z into COP2 control registers.
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
* RT22 is in BOTH $2.high AND $4.low (shared bit position). OP reads from $2.high.
@@ -214,121 +233,101 @@ internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa,
* The $2 and $4 writes don't clobber each other (separate registers).
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
* This is libpsyx's OuterProduct12 convention EXACTLY. */
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
* This is libpsyx's OuterProduct12 convention. */
gte_mv_to_ctrl_r(r.b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
gte_mv_to_ctrl_r(r.c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
/* Load uz into the RT diagonal. */
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
gte_mv_to_ctrl_r(r.a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
GteDelay_ nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
/* Load up_in into IR (the second operand for OP). */
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = up_in.x */
gte_mv_to_data_r(r.d, C2_IR1), /* IR1 = up_in.x */
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = up_in.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_mv_to_data_r(r.t0, C2_IR3), /* IR3 = up_in.z */
GteDelay_ nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product, /* OP: MAC1/2/3 = uz × up_in
gte_cmdw_cross, /* OP: MAC1/2/3 = uz × up_in
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
* MAC2 = IR1*D3 - IR3*D1 = up_in.x*uz.z.high - up_in.z*uz.x
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
* For up_in = (0, -fp_one, 0):
* MAC1 = 0 - (-fp_one)*uz.z.high = fp_one*uz.z.high
* MAC2 = 0 - 0 = 0
* MAC3 = (-fp_one)*uz.x - 0 = -fp_one*uz.x */
/* Restore the RT slots we clobbered. */
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
gte_mv_to_ctrl_r(r.target0, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
gte_mv_to_ctrl_r(r.target1, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop, /* MFC2 retirement */
mac_gte_mv_from_data_r_mac123(r.a, r.b, r.c),
GteDelay_ nop, /* MFC2 retirement */
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
shift_aright(r_a, r_a, 12),
shift_aright(r_b, r_b, 12),
shift_aright(r_c, r_c, 12),
mac_shift_aright_v3_self(r.a, r.b, r.c, 12),
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_store_word_v3(r.a, r.b, r.c, r.f, 0),
mac_yield()
})
typedef Struct_(RegUse_resolve_look_at__cross_uz_ux_to_up_proc) {
Reg const scratch; /* pinned T4 */
Reg_(V3_S4) a; /* uz components, then MAC / out */
Reg_(V3_S4) b; /* ux components */
union { Reg t0, up; }; /* &up, dedicated */
union { Reg t1, uz, rt11; }; /* &uz, then RT11 save */
union { Reg t2, ux, rt22; }; /* &ux, then RT22 save */
};
/* Atom 4: cross uz × ux → up. */
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
, U4 r_d /* load b.x */
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa,
RegUse_resolve_look_at__cross_uz_ux_to_up_proc r
) MipsAtom_Proc_(aa, {
/* Compute the three scratch pointers from r_scratch. */
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
nop,
add_si(r.uz, r.scratch, O_(ResolveLookAtScratch,uz)), /* r.uz = &uz */
add_si(r.ux, r.scratch, O_(ResolveLookAtScratch,ux)), /* r.ux = &ux */
add_si(r.up, r.scratch, O_(ResolveLookAtScratch,up)), /* r.up = &up (out) */
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
load_word(r_a, r_g, O_(V3_S4,x)),
load_word(r_b, r_g, O_(V3_S4,y)),
load_word(r_c, r_g, O_(V3_S4,z)),
nop,
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
load_word(r_d, r_h, O_(V3_S4,x)),
load_word(R_AT, r_h, O_(V3_S4,y)),
load_word(R_V0, r_h, O_(V3_S4,z)),
nop,
mac_load_v3s4(r.a, r.uz, 0), LdSlot_
mac_load_v3s4(r.b, r.ux, 0), LdSlot_ /* taken by gte_mv_from_ctrl_r */
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
* ctc2 RT restore. */
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux, ctc2 RT restore. */
/* Save the two RT control-register slots OP will clobber (reusing
* r_g/r_h — they're no longer needed as scratch pointers). */
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
/* Save the two RT control-register slots OP will clobber (reusing r.uz/r.ux — they're no longer needed as scratch pointers). */
gte_mv_from_ctrl_r(r.rt11, gte_cr_RT11), /* r.rt11 = C2 $0 (RT11|RT12) */
gte_mv_from_ctrl_r(r.rt22, gte_cr_RT22), /* r.rt22 = C2 $4 (RT22|RT33) */
/* Load uz into the RT diagonal — same packing as atom 1.
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4).
* OP reads RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high. */
gte_mv_to_ctrl_r(r.a.y, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
gte_mv_to_ctrl_r(r.a.z, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
gte_mv_to_ctrl_r(r.a.x, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
GteDelay_ nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
/* Load ux into the IR registers (the second operand for OP). */
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_mv_to_data_r(r.b.x, C2_IR1), /* IR1 = ux.x */
gte_mv_to_data_r(r.b.y, C2_IR2), /* IR2 = ux.y */
gte_mv_to_data_r(r.b.z, C2_IR3), /* IR3 = ux.z */
GteDelay_ nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
gte_cmdw_outer_product,
gte_cmdw_cross,
/* Restore the RT slots we clobbered. */
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
gte_mv_to_ctrl_r(r.rt11, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
gte_mv_to_ctrl_r(r.rt22, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
mac_gte_mv_from_data_r_mac123(r.a.x, r.a.y, r.a.z),
GteDelay_ nop,
gte_mv_from_data_r(r_a, C2_MAC1),
gte_mv_from_data_r(r_b, C2_MAC2),
gte_mv_from_data_r(r_c, C2_MAC3),
nop,
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
shift_aright(r_a, r_a, 12),
shift_aright(r_b, r_b, 12),
shift_aright(r_c, r_c, 12),
store_word(r_a, r_f, O_(V3_S4,x)),
store_word(r_b, r_f, O_(V3_S4,y)),
store_word(r_c, r_f, O_(V3_S4,z)),
mac_shift_aright_v3_self(r.a.x, r.a.y, r.a.z, 12),
mac_store_v3s4(r.a, r.up, 0),
mac_yield()
})
@@ -336,165 +335,81 @@ internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
};
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye).
*
* GPR codes (assigned by resolve_look_at_init):
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
*
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
*
* Struct layout (per duffle/math.h):
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
*
* Translation column: GTE MVMVA with the world rotation matrix pre-set
* (helper emits set_gte_world before the bundle, per the bundle design).
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
typedef Struct_(RegUse_resolve_look_at__populate_proc) {
Reg const scratch;
Reg look_at;
Reg_(V3_S4) row; /* one matrix row, reused */
Reg ux;
Reg uy;
Reg uz;
};
/* Atom 6a: write look_at->m[][] from ux/uy/uz as S2. Zero t[].
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; }
* m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
*/
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
, U4 r_look_at
, U4 r_scratch
, U4 r_pux, U4 r_puy, U4 r_puz
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa,
RegUse_resolve_look_at__populate_proc r
) MipsAtom_Proc_(aa, {
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
nop,
add_si(r.ux, r.scratch, O_(ResolveLookAtScratch,ux)), /* r.ux = &ux */ LdSlot_
add_si(r.uy, r.scratch, O_(ResolveLookAtScratch,uy)), /* r.uy = &uy */ LdSlot_
add_si(r.uz, r.scratch, O_(ResolveLookAtScratch,uz)), /* r.uz = &uz */ LdSlot_
/* ── m[0] = (S2)ux ── */
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
/* ── m[1] = (S2)uy ── */
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
/* ── m[2] = (S2)uz ── */
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
nop,
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
mac_load_v3s4(r.row, r.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4,m[0])),
mac_load_v3s4(r.row, r.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4,m[1])),
mac_load_v3s4(r.row, r.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4,m[2])),
/* Zero t[0..2] — atom 6c writes the final values here. */
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
mac_store_v3s4(v3s4_R_0(), r.look_at, O_(MT3_S2S4,t)),
mac_yield()
})
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
* Stores off to scratch+96 (overwriting the packed pos).
typedef Struct_(RegUse_resolve_look_at__matrix_vector_proc) {
Reg const scratch;
Reg look_at;
Reg eye; /* &scratch.eye; store dest for off */
Reg_(V3_S4) v; /* RT words, then -eye, then off */
};
/* Atom 6b: off = look_at.m * (-eye) >> 12. Stores off over scratch.eye.
*
* GPR codes (assigned by resolve_look_at_init):
* r_scratch : R_ResolveScratch (R_T4) — scratch base
* r_peye : pointer to eye (slot +96, reused as off destination)
* r_tmp0/1/2: -eye + GTE transfer scratch
*
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
* C11 ApplyMatrixLV:
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
* 2. lw v.x/y/z from memory
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3)
* 5. mfc2 MACs
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
* 7. mfc2 MACs
* 8. Combine: (pass1 << 3) + pass2
*/
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
, U4 r_scratch
, U4 r_peye
, U4 r_look_at
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa,
RegUse_resolve_look_at__matrix_vector_proc r
) MipsAtom_Proc_(aa, {
/* === EXACT C11 ApplyMatrixLV replication ===
* The C11 does:
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
* 2. lw v.x/y/z from memory
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3)
* 5. mfc2 MACs
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
* 7. mfc2 MACs
* 8. Combine: (pass1 << 3) + pass2
*
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
* The combine simplifies: result = 0 + pass2 = pass2.
* So we skip the S15 decomposition and just do pass 2 directly.
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* Pop look_at* from tape. */
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* Load RT from look_at.m into C2[0..4]. Packed S2 pairs, same as set_gte_mt3s2s4. */
load_word( r.v.x, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_ add_si(r.eye, r.scratch, O_(ResolveLookAtScratch,eye)), /* r.eye = &eye */
load_word( r.v.y, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.v.x, gte_cr_RT11),
load_word( r.v.z, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.v.y, gte_cr_RT12),
load_word( r.v.x, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.v.z, gte_cr_RT13),
load_half_u(r.v.y, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.v.x, gte_cr_RT21),
/* pos = -eye. The three loads also retire the last CTC2. */ gte_mv_to_ctrl_r(r.v.y, gte_cr_RT22),
GteDelay_ mac_load_p3s4(r.v, r.eye, 0), LdSlot_ mac_sub_v3s4(r.v, v3s4_R_0(), r.v), /* pos.x = -eye.x */
/* mtc2 pos (as S16) to IR1/2/3. The GTE takes low 16 bits. pos fits in S16. For negative pos, the 32-bit sign-extended value's low 16 bits = correct S16. */
gte_mv_to_data_r(r.v.x, C2_IR1),
gte_mv_to_data_r(r.v.y, C2_IR2),
gte_mv_to_data_r(r.v.z, C2_IR3),
GteDelay_ nop2,
/* r_peye = &eye (slot +96, reused as off destination). */
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
nop,
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
/* Load pos = -eye after the matrix load releases r_tmp0. */
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
nop,
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
sub_u(r_tmp1, R_0, r_tmp1),
sub_u(r_tmp2, R_0, r_tmp2),
/* === mtc2 pos (as S16) to IR1/2/3 ===
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
* 32-bit sign-extended value's low 16 bits = correct S16. */
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
* gives the correct S16 value (sign bit preserved). */
/* r_tmp0/1/2 already have pos values. */
gte_mv_to_data_r(r_tmp0, C2_IR1),
gte_mv_to_data_r(r_tmp1, C2_IR2),
gte_mv_to_data_r(r_tmp2, C2_IR3),
nop2, /* MTC2 retirement (2 slots) */
/* === MVMVA pass 2 — C11 ApplyMatrixLV command ===
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
gte_cmdw_mvmva_c11_pass2,
nop, /* GTE interlock */
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
gte_mv_from_data_r(r_tmp0, C2_MAC1),
gte_mv_from_data_r(r_tmp1, C2_MAC2),
gte_mv_from_data_r(r_tmp2, C2_MAC3),
nop,
/* === Store off → scratch+96 (overwriting pos) === */
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
gte_cmdw_mvmva_c11_pass2, GteDelay_ nop,
mac_gte_mv_from_data_r_mac123(r.v.x, r.v.y, r.v.z), GteDelay_ nop,
mac_store_v3s4(r.v, r.eye, 0),
mac_yield()
})
@@ -514,19 +429,15 @@ I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa
, U4 r_look_at, U4 r_scratch, U4 r_off_ptr
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
) MipsAtom_Proc_(aa, {
/* Pop look_at* from tape. */
// load_word(r_Vlook_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
// add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)),
nop,
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)), nop,
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */
mac_trans_mt3s3s4(r_look_at, r_off_ptr, r_tmp0, r_tmp1, r_tmp2),
mac_yield()
})
#pragma endregion resolve_look_at
#pragma endregion Atom Procs
@@ -658,15 +569,15 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
/* Load pad[0].buttons into R_T0. */
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
// Note(Ed): Potential op with delay slot?
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, 30),
add_si( R_T3, R_T3, 5),
@@ -675,8 +586,8 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
atom_label(exit_dpad_left)
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
add_si( R_T4, R_T4, -30),
add_si( R_T3, R_T3, -5),
@@ -686,7 +597,7 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
/* Analog left-stick X: dead zone 0x70..0x90.
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
@@ -695,14 +606,14 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
atom_label(dead_check_upper)
/* left_x >= 0x70 → check upper bound. */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
add_ui( R_T4, R_0, PadDeadZone_HighBound),
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
mac_yield_load(),
BdSlot_ mac_yield_load(), LdSlot_
atom_label(dead_low_active)
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
@@ -713,18 +624,18 @@ atom_label(dead_low_active)
/* R_T4 = cube_delta */
shift_aright(R_T4, R_T3, 2),
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
jump_rel(atom_offset(end_low, exit_stick)),
mac_yield_load(),
BdSlot_ mac_yield_load(), LdSlot_
atom_label(dead_high_active)
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
@@ -734,18 +645,18 @@ atom_label(dead_high_active)
/* delta = 0x80 - left_x (signed negative). */
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
shift_aright(R_T4, R_T3, 5),
add_u( R_T0, R_T0, R_T4),
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
atom_label(no_jump_fallthrough)
mac_yield_load(),
mac_yield_load(), LdSlot_
atom_label(exit_stick)
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
@@ -767,41 +678,41 @@ internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
load_word(R_T1, R_Cam, O_(Camera,pos.x)),
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ mac_yield_load(), LdSlot_
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_left_x)
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), BdSlot_ nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
atom_label(exit_right_x)
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
load_word(R_T1, R_Cam, O_(Camera,pos.y)), LdSlot_
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), BdSlot_ nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_up_y)
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), BdSlot_ nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
atom_label(exit_down_y)
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
load_word(R_T1, R_Cam, O_(Camera,pos.z)), LdSlot_
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), BdSlot_ nop,
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_cross_z)
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), BdSlot_ nop,
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
atom_label(exit_circle_z)
@@ -833,7 +744,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
mac_yield()
};
@@ -846,20 +757,20 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
GteDelay_ nop, gte_cmdw_rotate_translate_perspective_triple,
gte_cmdw_nclip,
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ nop,
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
* harmless because the OT entry that points to this prim is created later. */
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
mac_gte_store_g4_p012(R_PrimCursor),
@@ -871,7 +782,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
add_ui( R_AT, R_0, OrderingTbl_Len),
set_lt_u( R_AT, R_T1, R_AT),
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
mac_format_g4_color(R_PrimCursor,
/* c0 magenta */ 0xFF, 0x00, 0xFF,
@@ -903,7 +814,7 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
mac_yield()
};
@@ -950,7 +861,7 @@ internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitive
, atom_writes(R_TapePtr)
){
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
/* Calculate byte offset and store directly back to RAM */
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
+131 -121
View File
@@ -32,7 +32,7 @@
#pragma region Duffle TUs
#include "duffle/pad.c"
#include "duffle/math.atom.c"
#include "duffle/math.atom.h"
#include "duffle/mips.atom.c"
#include "duffle/gte.atom.c"
#include "duffle/gp.atom.c"
@@ -60,8 +60,12 @@ enum {
enum {
Scratchpad_Len = 1024,
MemTape_Len = 512,
ResolveLookAtArena_Words = 1024,
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
CT_InitAtomMem_Words = Kilo_(4),
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
};
typedef Struct_(SMemory) {
PrimitiveArena primitives;
@@ -85,8 +89,14 @@ typedef Struct_(SMemory) {
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
U4_V scratchpad; // d-cache
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
MipsAtom* normalize_v3s4;
// TODO(Ed): Convert normalize_v3s4 to a generic atom?
// This would allow us to reduce specializations with the loss being some cycles to loading registers.
// The cost would be 3 loads (scratch, src_ptr, dst_offset) from tape and
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
MipsAtom* resolve_look_at_atom_addrs[10];
MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
};
global SMemory smem;
extern SMemory smem;
@@ -131,158 +141,155 @@ I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4*
}
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
* share the GENERIC normalize_v3s4_proc from gte.atom.c
* 0: resolve_look_at__input_and_sub_proc
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
* 4: resolve_look_at__cross_uz_ux_to_up_proc
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
* 6: resolve_look_at__populate_and_translate_proc
*/
internal void resolve_look_at_init(void) {
internal void compile_resolve_look_at(void) {
/* Wrap the static arena in a MipsAtomBuilder. */
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_atom_addrs));
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_bundle));
U4 pin_mask = regfile_abi_mask | (1 << R_ResolveScratch);
RegFile rf = regfile(pin_mask);
#define ralloc() regfile_alloc(& rf)
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
U4 r_target_ptr = regfile_alloc(& rf);
U4 r_eye_ptr = regfile_alloc(& rf);
U4 r_up_in_ptr = regfile_alloc(& rf);
U4 r_tmp0 = regfile_alloc(& rf);
U4 r_tmp1 = regfile_alloc(& rf);
U4 r_tmp2 = regfile_alloc(& rf);
U4 r_tmp3 = regfile_alloc(& rf);
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
R_ResolveScratch,
r_target_ptr, r_eye_ptr, r_up_in_ptr,
r_tmp0, r_tmp1, r_tmp2, r_tmp3);
tb_emit_(AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
RegUse_(resolve_look_at_input_and_sub) {
.scratch = R_ResolveScratch,
.target = ralloc(),
.eye = ralloc(),
.up_in = ralloc(),
.t0 = ralloc(),
.t1 = ralloc(),
.t2 = ralloc(),
.t3 = ralloc(),
.t4 = ralloc(),
}
));
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 1: normalize fwd→uz === */
U2 src_offset = O_(ResolveLookAtScratch, fwd);
U2 dst_offset = O_(ResolveLookAtScratch, uz);
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
smem.resolve_look_at_bundle[1] = build_normalize_v3s4(& ab,
src_offset, dst_offset, RegUse_(build_normalize_v3s4){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
.src_ptr = ralloc(),
.dst_ptr = ralloc(),
.recip_est = ralloc(),
.norm = ralloc(),
.shift = ralloc(),
.src_x = ralloc(),
.t3 = ralloc(),
.t4 = ralloc(),
.t5 = ralloc(),
});
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 2: cross uz×up_in→right === */
U4 r_a_2 = R_T0;
U4 r_b_2 = R_T1;
U4 r_c_2 = R_T2;
U4 r_d_2 = R_T3;
U4 r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
U4 r_g_2 = R_T6; /* a ptr = scratch+16 */
U4 r_h_2 = R_T7; /* b ptr = scratch+128 */
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
R_ResolveScratch,
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
// smem.resolve_look_at_bundle[2] = AtomBundleEntry_(resolve_look_at,cross_uz_up_to_right)(& ab,
smem.resolve_look_at_bundle[2] = resolve_look_at_cross_uz_up_into_right(& ab,
RegUse_(resolve_look_at_cross_uz_up_into_right) {
.scratch = R_ResolveScratch,
.a = ralloc(),
.b = ralloc(),
.c = ralloc(),
.d = ralloc(),
.f = ralloc(),
.t1 = ralloc(),
.t2 = ralloc(),
.t0 = ralloc(),
});
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 3: normalize right→ux === */
src_offset = O_(ResolveLookAtScratch, right);
dst_offset = O_(ResolveLookAtScratch, ux);
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
smem.resolve_look_at_bundle[3] = build_normalize_v3s4(& ab,
src_offset, dst_offset, RegUse_(build_normalize_v3s4){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
.src_ptr = ralloc(),
.dst_ptr = ralloc(),
.recip_est = ralloc(),
.norm = ralloc(),
.shift = ralloc(),
.src_x = ralloc(),
.t3 = ralloc(),
.t4 = ralloc(),
.t5 = ralloc(),
});
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 4: cross uz×ux→up === */
U4 r_a_4 = R_T0;
U4 r_b_4 = R_T1;
U4 r_c_4 = R_T2;
U4 r_d_4 = R_T3;
U4 r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
U4 r_g_4 = R_T6; /* a ptr = scratch+16 */
U4 r_h_4 = R_T7; /* b ptr = scratch+48 */
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
R_ResolveScratch,
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
smem.resolve_look_at_bundle[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
RegUse_(resolve_look_at__cross_uz_ux_to_up_proc){
.scratch = R_ResolveScratch,
.a = ralloc_v3(), /* T0 T1 T2 */
.b = ralloc_v3(), /* T3 T5 T6 */
.t0 = ralloc(), /* T7 = up */
.t1 = ralloc(), /* V0 = uz / rt11 */
.t2 = ralloc(), /* V1 = ux / rt22 */
});
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 5: normalize up→uy === */
src_offset = O_(ResolveLookAtScratch, up);
dst_offset = O_(ResolveLookAtScratch, uy);
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
smem.resolve_look_at_bundle[5] = build_normalize_v3s4(& ab,
src_offset, dst_offset,
RegUse_(normalize_v3s4_proc){
RegUse_(build_normalize_v3s4){
.scratch = R_ResolveScratch,
.src_ptr = R_T0,
.dst_ptr = R_T1,
.recip_est = R_T6,
.norm = R_T7,
.shift = R_V0,
.src_x = R_T2,
.t3 = R_T3,
.t4 = R_T5,
.t5 = R_V1,
.src_ptr = ralloc(),
.dst_ptr = ralloc(),
.recip_est = ralloc(),
.norm = ralloc(),
.shift = ralloc(),
.src_x = ralloc(),
.t3 = ralloc(),
.t4 = ralloc(),
.t5 = ralloc(),
});
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
U4 r_look_at_6a = R_T0; /* tape pop → look_at* */
U4 r_scratch_6a = R_ResolveScratch;
U4 r_pux_6a = R_T1;
U4 r_puy_6a = R_T3;
U4 r_puz_6a = R_T5;
U4 r_tmp0_6a = R_T2;
U4 r_tmp1_6a = R_T6;
U4 r_tmp2_6a = R_V0;
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
r_look_at_6a, r_scratch_6a,
r_pux_6a, r_puy_6a, r_puz_6a,
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
smem.resolve_look_at_bundle[6] = resolve_look_at__populate_proc(& ab,
RegUse_(resolve_look_at__populate_proc){
.scratch = R_ResolveScratch,
.look_at = ralloc(), /* T0 */
.row = ralloc_v3(), /* T1 T2 T3 */
.ux = ralloc(), /* T5 = ux */
.uy = ralloc(), /* T6 = uy */
.uz = ralloc(), /* T7 = uz */
});
regfile_reset_to_mask(& rf, pin_mask);
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
* the matrix pointer (popped from tape). It does NOT need GPR
* assignment from us — it has its own internal GPR usage.
* We just take its address. */
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
smem.resolve_look_at_bundle[7] = (MipsAtom*) & set_gte_mt3s2s4;
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
* Uses mac_apply_matrix_lv component macro which internally uses
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
* for the mfc2/store. We pass our GPRs. */
U4 r_scratch_6b = R_ResolveScratch;
U4 r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
U4 r_look_at_6b = R_T0; /* tape pop → look_at* */
U4 r_tmp0_6b = R_T2;
U4 r_tmp1_6b = R_T3;
U4 r_tmp2_6b = R_T5;
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
r_scratch_6b, r_peye_6b, r_look_at_6b,
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) === */
smem.resolve_look_at_bundle[8] = resolve_look_at__matrix_vector_proc(& ab,
RegUse_(resolve_look_at__matrix_vector_proc){
.scratch = R_ResolveScratch,
.look_at = ralloc(), /* T0 */
.eye = ralloc(), /* T1 */
.v = ralloc_v3(), /* T2 T3 T5 */
});
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
U4 r_look_at_6c = R_T0; /* tape pop → look_at* */
U4 r_scratch_6c = R_ResolveScratch;
U4 r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
U4 r_tmp0_6c = R_T2;
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
smem.resolve_look_at_bundle[9] = resolve_look_at__trans_matrix_proc(& ab,
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c, R_T3, R_T4);
/* Sanity check: arena didn't overflow. */
assert(ab.used <= ResolveLookAtArena_Size);
#undef ralloc
}
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
@@ -296,36 +303,41 @@ internal void resolve_look_at_init(void) {
* ----
* 5 tb_data words total per frame.
*/
I_ void resolve_look_at(
TapeBuilder_R tb
I_ void resolve_look_at(TapeBuilder_R tb
, MT3_S2S4* look_at
, P3_S4* eye
, P3_S4* target
, V3_S4* up_in
){
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
tb_emit(tb, smem.resolve_look_at_bundle[0]); {
tb_data(tb, u4_(target));
tb_data(tb, u4_(eye));
tb_data(tb, u4_(up_in));
tb_data(tb, u4_(smem.scratchpad));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
tb_emit(tb, smem.resolve_look_at_bundle[1]); {
// tb_data(tb, u4_(Scratchpad_Loc));
}
tb_emit(tb, smem.resolve_look_at_bundle[2]); { }
tb_emit(tb, smem.resolve_look_at_bundle[3]); {
// tb_data(tb, u4_(Scratchpad_Loc));
}
tb_emit(tb, smem.resolve_look_at_bundle[4]); { }
tb_emit(tb, smem.resolve_look_at_bundle[5]); {
// tb_data(tb, u4_(Scratchpad_Loc));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
tb_emit(tb, smem.resolve_look_at_bundle[6]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
tb_emit(tb, smem.resolve_look_at_bundle[7]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
tb_emit(tb, smem.resolve_look_at_bundle[8]); {
tb_data(tb, u4_(look_at));
}
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
tb_emit(tb, smem.resolve_look_at_bundle[9]); {
// tb_data(tb, u4_(look_at));
}
}
@@ -519,8 +531,7 @@ int main(void)
/* Direct BIOS: poll both ports during VBlank. */
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
resolve_look_at_init();
compile_resolve_look_at();
/* Pinned registers for the GPU init atom. */
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
@@ -541,4 +552,3 @@ int main(void)
return 0;
}
GCC_OPTIMIZATION_ENABLE
+137 -23
View File
@@ -1014,14 +1014,20 @@ end
-- Section 7: domain tables
-- ════════════════════════════════════════════════════════════════════════════
-- The annotation DSL has been reduced to a single annotation macro: atom_info(atom_bind(Binds_X), atom_reads(...), atom_writes(...))
-- All phase / region / cadence / async / resource / group tokens have been dropped.
-- They may be reintroduced later as optional sub-calls of atom_info;
-- For now, the parser only recognizes atom_info + its three sub-calls (atom_bind, atom_reads, atom_writes).
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
M.TAPE_ATOM_MACROS = {
["atom_info"] = { kind = "info", binds = false },
}
-- Empty C macros that prefix the next encoder. Zero words.
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
M.DELAY_MARKERS = {
["GteDelay_"] = true,
["LdSlot_"] = true,
["BdSlot_"] = true,
["DmaSlot_"] = true,
}
-- GTE command-alias resolution table.
--
-- Maps every source-side GTE command macro to its canonical short ident.
@@ -1328,6 +1334,12 @@ M.GTE_CR_ALIAS_GROUPS = {
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
}
-- Packed RT slots named by the gte.h packed-slot comment.
-- first must be written before second.
M.GTE_PACKED_SLOT_RELATIONS = {
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
}
-- Operand-class table for the COP2->GPR load-delay check.
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
@@ -2236,6 +2248,43 @@ local function _project_emission_inner(root_body_entry, ctx_table)
local invocation_stack = {} -- stack of currently-open invocation records
local next_inv_id = 0
local reg_use_schema = ctx_table.reg_use_schema
local reg_use_param = ctx_table.reg_use_param
local atom_name = ctx_table.atom_name
local slot_readonly = {}
if reg_use_schema then
for _, slot in ipairs(reg_use_schema.slots or {}) do
slot_readonly[slot.name] = slot.readonly == true
end
end
local function apply_sub(sub_map, operand)
if not (sub_map and type(operand) == "string") then return operand end
if sub_map[operand] then return sub_map[operand] end
local dot = operand:find(".", 1, true)
if dot then
local head = operand:sub(1, dot - 1)
local mapped = sub_map[head]
if type(mapped) == "string" then
return mapped .. operand:sub(dot)
end
end
return operand
end
local function resolve_gpr_key(operand)
if type(operand) ~= "string" then return nil end
if operand:sub(1, 2) == "R_" then return operand end
if not (reg_use_schema and reg_use_param) then return nil end
local prefix = reg_use_param .. "."
if operand:sub(1, #prefix) ~= prefix then return nil end
local member_path = operand:sub(#prefix + 1)
local slot = reg_use_schema.alias_to_slot[member_path]
if not slot then return nil, member_path end
return "reguse:" .. atom_name .. ":" .. slot, nil, slot
end
local function open_invocation_ids_snapshot()
local ids = {}
for _, inv in ipairs(invocation_stack) do
@@ -2246,7 +2295,7 @@ local function _project_emission_inner(root_body_entry, ctx_table)
local function emit_word(encoder, args, line, word_call_text,
def_source_now, def_line_now,
immediate_call_text, root_call_text_w)
immediate_call_text, root_call_text_w, sub_map)
local inv_ids = open_invocation_ids_snapshot()
local outermost = inv_ids[1] or 0
-- For words emitted at the root atom body, `immediate_call_text` is nil and the walker's `word_call_text` (the word's own token, e.g. "nop") becomes the effective call_text.
@@ -2254,6 +2303,42 @@ local function _project_emission_inner(root_body_entry, ctx_table)
-- The call that triggered the body expansion we're currently walking.
local eff_call_text = immediate_call_text or word_call_text
local eff_root_call_text = root_call_text_w
local gpr_keys = nil
if reg_use_schema or sub_map then
gpr_keys = {}
for pos, arg in ipairs(args or {}) do
local effective = apply_sub(sub_map, arg)
local key, unresolved, slot = resolve_gpr_key(effective)
gpr_keys[pos] = key
if unresolved then
errors[#errors + 1] = {
kind = "reguse_unresolved",
line = line,
msg = string.format("RegUse operand %q does not resolve in schema %q",
effective, (reg_use_schema and reg_use_schema.name) or "?"),
}
end
if key and slot and slot_readonly[slot] then
local effects = M.INSTRUCTION_GPR_EFFECTS or {}
local row = effects[encoder]
if row and row.writes then
for _, wpos in ipairs(row.writes) do
if wpos == pos then
errors[#errors + 1] = {
kind = "reguse_const_write",
line = line,
msg = string.format("RegUse slot %q is Reg const; %s writes it",
slot, encoder),
}
end
end
end
end
end
end
if not reg_use_schema then
gpr_keys = nil
end
items[#items + 1] = {
kind = "word",
encoder = encoder,
@@ -2265,6 +2350,7 @@ local function _project_emission_inner(root_body_entry, ctx_table)
root_call_text = eff_root_call_text,
invocation_ids = inv_ids,
outermost_invocation_id = outermost,
gpr_keys = gpr_keys,
}
word_events[#word_events + 1] = {
i = word_idx,
@@ -2277,6 +2363,7 @@ local function _project_emission_inner(root_body_entry, ctx_table)
invocation_ids = inv_ids,
outermost_invocation_id = outermost,
word_count = 1,
gpr_keys = gpr_keys,
}
word_idx = word_idx + 1
end
@@ -2380,6 +2467,15 @@ local function _project_emission_inner(root_body_entry, ctx_table)
pos = (next_pos > pos) and next_pos or (pos + 1)
goto continue_loop
end
if M.DELAY_MARKERS[ident] then
local arg_pos = nil
if consuming_encoder and consuming_paren then
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
end
emit_marker("delay", ident, nil, tok_line, nil, nil, consuming_encoder, arg_pos)
pos = after
goto continue_loop
end
if ident ~= "atom_label" and ident ~= "atom_offset" then
-- Ordinary ident; nothing to emit, step past the ident only.
pos = after
@@ -2520,14 +2616,24 @@ local function _project_emission_inner(root_body_entry, ctx_table)
local line_of = body_entry.line_of or M.LineIndex("")
local def_source = body_entry.source or ""
local def_line = body_entry.declaration or 0
local sub_map = body_entry.sub_map
-- Per-token dispatch: each matched branch returns; only the fall-through
-- "opaque word" emit handles direct encoders + mac_X-without-component.
local function process_token(bt)
local tok = M.trim(bt.tok or "")
if tok == "" then return end
local ident = M.read_ident(tok, 1) or "?"
local ident, after = M.read_ident(tok, 1)
if not ident then ident = "?" end
local _, args = token_ident_and_args(tok)
local tok_line = line_of(body_off + bt.rel) or 0
if M.DELAY_MARKERS[ident] then
emit_marker("delay", ident, nil, tok_line)
local rest = M.trim(tok:sub(after or (#tok + 1)))
if rest ~= "" then
process_token({ tok = rest, rel = bt.rel })
end
return
end
-- embedded markers live only in non-marker tokens.
-- Pass `ident` as the consuming instruction so `emit_embedded_markers` can compute each marker's arg position + record the consuming_encoder for the offsets pass.
-- Canonicalize `jump_rel` to `branch_equal` (its preprocessor-expanded form) so the `consuming_encoder` metadata in marker records is canonical.
@@ -2575,12 +2681,22 @@ local function _project_emission_inner(root_body_entry, ctx_table)
-- Propagate trackers into the recursive walk:
-- immediate_call_text = this call's tok (the IMMEDIATE outer call for words emitted in this body)
-- root_call_text = the OUTERMOST call (immutable across the recursion)
local formal_names = ctx_table.component_index[bare]
and ctx_table.component_index[bare].arg_names
local child_map = nil
if formal_names then
child_map = {}
for i, fname in ipairs(formal_names) do
child_map[fname] = apply_sub(sub_map, args[i])
end
end
walk_body_entry({
body_tokens = comp.body_tokens or {},
body_off = comp.body_off or 0,
line_of = comp.line_of,
source = comp.source,
declaration = comp.declaration,
sub_map = child_map,
},
inv.id,
invocation_root_call_text,
@@ -2618,7 +2734,7 @@ local function _project_emission_inner(root_body_entry, ctx_table)
local n = resolve_count(ident, tok_line)
local out_ident = (ident == "nop2") and "nop" or ident
for _ = 1, n do
emit_word(out_ident, args, tok_line, tok, def_source, def_line, walk_immediate_call_text, walk_root_call_text)
emit_word(out_ident, args, tok_line, tok, def_source, def_line, walk_immediate_call_text, walk_root_call_text, sub_map)
end
end
@@ -2665,6 +2781,7 @@ end
--- `nop2` is normalized to encoder `nop` (per the spec).
--- * `atom_label(F)` markers: one `label` item with `name = "F"`, `word_index = current word_idx`; zero-width (does NOT advance word_idx).
--- * `atom_offset(B, T)` markers: one `offset` item with `name = "B"`, `target = "T"`, `word_index = current word_idx`; zero-width.
--- * Delay markers (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`): one `delay` item; zero-width. The following encoder is the next token.
--- * `mac_X(...)` calls: emit `invoke_begin` (zero-width), recurse into the component body, emit `invoke_end` (zero-width).
--- The component body's words land between the begin/end pair; one invocation record is allocated per call (monotonic ID per atom).
--- * Unknown uncounted macros emit 1 opaque word + one warning per occurrence.
@@ -2683,7 +2800,7 @@ end
--- @param components table -- bare-name → component definition (corpus.components); REQUIRED — consumed at the invocation-construction site to stamp
--- `invocation.debug_skip`. A missing or non-table `components` raises a fail-loud error rather than silently falling back.
--- @return EmissionProjection
function M.project_emission(body_text, component_index, word_counts, components)
function M.project_emission(body_text, component_index, word_counts, components, reg_use_ctx)
-- The recursive walk delegates to `_project_emission_inner` so component bodies (which arrive as
-- `{body_tokens, body_off, line_of, source, declaration}` records from `corpus.component_body_index`)
-- re-enter the same walker with the same shared output state.
@@ -2726,6 +2843,10 @@ function M.project_emission(body_text, component_index, word_counts, components)
component_index = component_index or {},
word_counts = word_counts or {},
components = components,
reg_use_schema = reg_use_ctx and reg_use_ctx.reg_use_schema,
reg_use_param = reg_use_ctx and reg_use_ctx.reg_use_param,
atom_name = reg_use_ctx and reg_use_ctx.atom_name,
schema_name = reg_use_ctx and reg_use_ctx.schema_name,
})
end
@@ -2802,18 +2923,17 @@ end
-------------------------------------------------------------------------------
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
--
-- After the `sym` arg was dropped from MipsAtom_Proc_, the atom name is
-- derived from the preceding `MipsAtom* X_proc(args)` function declaration.
-- The atom name is the preceding `MipsAtom* ident(args)` function ident.
-- This function walks backward from `before_pos` to find it.
--
-- Returns (raw_name, args_inner) or (nil, nil).
-- raw_name — e.g. "normalize_v3s4" (the _proc suffix is stripped)
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
-- Returns (raw_name, args_inner, func_ident, after_paren) or (nil, nil).
-- raw_name — the function ident as written
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
-- after_paren — source position after the function `)`
--
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips
-- whitespace + qualifiers (internal, I_, FI_, comments) until it finds an
-- ident followed by "(". That ident is the function name (with _proc suffix);
-- the suffix is stripped to get raw_name. The parens contents are the args.
-- ident followed by "(". That ident is the name. The parens contents are the args.
-------------------------------------------------------------------------------
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
local search_pos = 1
@@ -2858,15 +2978,9 @@ function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
-- check if the next non-ws char after ident is "("
local next_pos = M.skip_ws_and_cmt(source, ident_end)
if source:sub(next_pos, next_pos) == "(" then
local inner = M.read_parens(source, next_pos)
local inner, after_paren = M.read_parens(source, next_pos)
if inner then
-- strip the _proc suffix to get the atom name
local proc_suffix = "_proc"
if #ident > #proc_suffix and ident:sub(-#proc_suffix) == proc_suffix then
return ident:sub(1, #ident - #proc_suffix), inner
end
-- no _proc suffix — return as-is
return ident, inner
return ident, inner, ident, after_paren
end
end
-- ident not followed by "(" — it's a qualifier; skip it
+2 -2
View File
@@ -612,13 +612,13 @@ function M.read_elf_sections(elf_path, section_names)
end
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
--- Returns a map `{name -> {addr, size_bytes}}` for every `code_<name>` symbol.
--- Returns a map `{name -> {addr, size_bytes}}` for every defined symbol.
---
--- **Conventions:**
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
--- - Keys are the ELF symbol names as written (the C ident).
--- - `st_size > 0` filter excludes undefined/imported symbols.
---
--- @param elf_path Path
+1 -1
View File
@@ -16,7 +16,7 @@ define tape_atoms
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
end
document tape_atoms
List every tape atom symbol in the loaded ELF (code_<name>) with its .rodata address and word count.
List every tape atom symbol in the loaded ELF with its .rodata address and word count.
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
end
+2 -2
View File
@@ -477,8 +477,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
local atoms = {}
for _, a in ipairs(scan.atoms) do
if a.kind == "atom" then
atoms[#atoms + 1] = { line = a.line, name = a.raw_name }
if a.kind == "atom" or a.kind == "atom_proc" then
atoms[#atoms + 1] = { line = a.line, name = a.raw_name or a.name }
end
end
+19 -7
View File
@@ -83,6 +83,7 @@ local function canonical_word_entries(atom)
line = event.call_line or item.line or 0,
text = event.call_text or item.call_text or "",
body_line = event.body_line or item.body_line or item.line or 0,
gpr_keys = event.gpr_keys,
invocation = (event.outermost_invocation_id
and paths.invocations
and paths.invocations[event.outermost_invocation_id]) or nil,
@@ -260,12 +261,12 @@ local function append_gdb_commands(lines, matched)
for _, a in ipairs(matched) do
-- gdb 12.1 quirk: literals in printf args require an attached target.
-- Use the per-atom convenience vars set above as printf args.
lines[#lines + 1] = string.format(' printf " code_%%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
a.idx, a.idx, a.idx)
end
lines[#lines + 1] = "end"
lines[#lines + 1] = "document tape_atoms"
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF (code_<name>) with .rodata addr + word count."
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
lines[#lines + 1] = "end"
lines[#lines + 1] = ""
@@ -284,10 +285,10 @@ local function append_gdb_commands(lines, matched)
for _, a in ipairs(matched) do
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
lines[#lines + 1] = string.format(' printf " Breakpoint set at code_%s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
lines[#lines + 1] = "end"
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
lines[#lines + 1] = string.format(" Set a breakpoint at code_%s.", a.name)
lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
lines[#lines + 1] = "end"
lines[#lines + 1] = ""
end
@@ -322,7 +323,7 @@ local function append_gdb_commands(lines, matched)
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
lines[#lines + 1] = string.format(' printf "atom: code_%%s\\n", $__atom_name_%d', a.idx)
lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
@@ -491,8 +492,19 @@ function M.render_atom_source_map(atom)
local lines = {}
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
for _, entry in ipairs(entries) do
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
local word_line = string.format("WORD %d LINE %d TEXT %s",
entry.pos, entry.line, entry.text)
local keys = {}
for pos = 1, 16 do
local k = entry.gpr_keys and entry.gpr_keys[pos]
if type(k) == "string" and k:sub(1, 7) == "reguse:" then
keys[#keys + 1] = k
end
end
if #keys > 0 then
word_line = word_line .. " KEYS " .. table.concat(keys, ",")
end
lines[#lines + 1] = word_line
end
lines[#lines + 1] = "ENDATOM"
return table.concat(lines, "\n") .. "\n"
@@ -529,7 +541,7 @@ function M.render_atom_provenance(atom, wc, rel_path)
return table.concat(lines, "\n") .. "\n"
end
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
--- Pass entry. For each source that declares at least one tape atom,
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
+27 -15
View File
@@ -7,7 +7,7 @@
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
---
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
--- the ELF symbol is the C ident. Raw `MipsCode code_*` is leftover, not the atom rule.
---
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
@@ -159,6 +159,14 @@ local function extract_arg_names(args_str)
return names
end
local function formal_arg_names(args_str)
local names = extract_arg_names(args_str)
if not names then return nil end
if names[1] == "ab" then table.remove(names, 1) end
if #names == 0 then return nil end
return names
end
-- ════════════════════════════════════════════════════════════════════════════
-- Component projection (read from pre-scanned SourceScan)
-- ════════════════════════════════════════════════════════════════════════════
@@ -179,7 +187,7 @@ local function project_components(source, scan)
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
-- `tb_emit` of the C ident, NOT inlined as a macro. Including `atom_proc` here
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
if a.kind == "comp_bare" or a.kind == "comp_proc" then
@@ -197,6 +205,7 @@ local function project_components(source, scan)
body_off = a.body_off,
body_tokens = a.body_tokens,
args = args,
arg_names = formal_arg_names(args),
comment = comment,
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
debug_skip = a.debug_skip == true,
@@ -384,6 +393,10 @@ end
--- @param cache table<string, integer>
--- @return integer
local function gp0_contrib_rec(name, comp_by_name, cache)
if name:match("^insert_ot_tag") then
cache[name] = 0
return 0
end
if cache[name] ~= nil then return cache[name] end
cache[name] = -1
local cc = comp_by_name[name]
@@ -399,8 +412,15 @@ local function gp0_contrib_rec(name, comp_by_name, cache)
-- Nested `mac_X(...)` call: recurse.
local nested = ident:sub(MAC_PREFIX_LEN + 1)
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
elseif ident == "gte_sw" then
n = n + 1
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
if trimmed:find("R_PrimCursor", 1, true) then
if trimmed:find("R_PrimCursor", 1, true)
or trimmed:find("O_(Poly_", 1, true)
or trimmed:find("r_prim_cursor", 1, true)
or trimmed:find("r_primitive_cursor", 1, true)
or trimmed:find("r_base", 1, true)
then
n = n + 1
end
end
@@ -466,18 +486,9 @@ end
--- @param args_str string|nil
--- @return string
local function signature_from_args(args_str)
local arg_names = extract_arg_names(args_str)
if arg_names and #arg_names > 0 then
-- Drop the leading `ab` (atom-builder) first arg if present.
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
if arg_names[1] == "ab" then
table.remove(arg_names, 1)
end
if #arg_names > 0 then
return table.concat(arg_names, ", ")
end
return "..." -- `ab` was the only arg; fall through to variadic
local names = formal_arg_names(args_str)
if names then
return table.concat(names, ", ")
end
return "..."
end
@@ -710,6 +721,7 @@ local function update_canonical_component_body_index(corpus, src, components, sc
source = src.path,
declaration = c.line,
kind = c.kind,
arg_names = c.arg_names,
}
end
end
+4 -3
View File
@@ -2,7 +2,7 @@
---
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
--- APPENDS synthetic DWARF line-program sequences for every `code_<name>` atom, EXTENDS the `.debug_aranges`
--- APPENDS synthetic DWARF line-program sequences for every tape atom, EXTENDS the `.debug_aranges`
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
--- existing main compilation unit in `.debug_info` (no second compilation unit).
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
@@ -1344,7 +1344,7 @@ local function build_new_abbrev()
attr( DW_AT_name, DW_FORM_string)
.. attr(DW_AT_low_pc, DW_FORM_addr)
.. attr(DW_AT_high_pc, DW_FORM_addr)
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; lets gdb's symbol-table lookup resolve to our subprogram (not the gcc global `code_<name>` const U4 array)
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; gdb resolves the subprogram, not the gcc global array
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
attr( DW_AT_name, DW_FORM_string)
@@ -1857,7 +1857,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
end
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
-- Subprogram names match nm symbols without a `code_` prefix.
-- Subprogram names match the written C ident (the ELF symbol).
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
for _, atom in ipairs(atom_table) do
@@ -2313,5 +2313,6 @@ end
M.compute_loclists_offsets_for_test = compute_loclists_offsets
M.build_debug_loclists_section_for_test = build_debug_loclists_section
M.tape_piece_size_for_test = tape_piece_size
M.build_atom_table_for_test = build_atom_table
return M
+21 -1
View File
@@ -155,8 +155,28 @@ local function project_atom(atom_record, src, corpus)
local body = atom_record.body or ""
local wc = corpus.word_counts or {}
local cbi = corpus.component_body_index or {}
local schema = nil
if atom_record.reg_use_schema_name then
schema = corpus.reg_use_schemas and corpus.reg_use_schemas[atom_record.reg_use_schema_name]
end
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
local proj = duffle.project_emission(body, cbi, wc, corpus.components)
local proj = duffle.project_emission(body, cbi, wc, corpus.components, {
reg_use_schema = schema,
reg_use_param = atom_record.reg_use_param_name,
atom_name = atom_record.name,
schema_name = atom_record.reg_use_schema_name,
})
if atom_record.reg_use_schema_name and not schema then
proj.errors[#proj.errors + 1] = {
kind = "reguse_missing_schema",
msg = string.format("RegUse schema %q is missing", atom_record.reg_use_schema_name),
}
end
for _, err in ipairs(corpus.reg_use_errors or {}) do
if err.schema_name == atom_record.reg_use_schema_name then
proj.errors[#proj.errors + 1] = err
end
end
local paths = {
tokens = atom_record.body_tokens or {},
line_in_body = duffle.build_body_line_index(body),
+2 -1
View File
@@ -1,7 +1,8 @@
--- passes/offsets.lua — Branch-offset generator.
---
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
--- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
---
+577 -226
View File
@@ -4,8 +4,8 @@
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
--- - `build/gen/annotation_validation.txt` — the project summary.
---
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass iterates the dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
--- The canonical `corpus.sources_by_dir` projection groups sources by directory.
--- This pass builds one ModuleView per directory and walks SECTION_RENDERERS.
-- ════════════════════════════════════════════════════════════════════════════
-- Module-scope requires + package.path setup
@@ -20,11 +20,6 @@
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
@@ -225,7 +220,7 @@ local function render_module_atoms_md(dir, dir_sources, wc)
for _, atom in ipairs(atoms_list) do
lines[#lines + 1] = string.format(
"### atom: %s (line %d, %d words)",
atom.name, atom.line or 0, #(atom.paths.items or {}))
atom.name, atom.line or 0, #((atom.paths or {}).word_events or {}))
lines[#lines + 1] = ""
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
lines[#lines + 1] = "```"
@@ -246,17 +241,544 @@ local function render_module_atoms_md(dir, dir_sources, wc)
return table.concat(lines, "\n") .. "\n"
end
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
--- Aggregates annotation + static-analysis content across all sources in `dir`.
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
--- @param dir string
--- @param dir_sources SourceFile[]
--- @param annot_results AnnotationResult[]
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
--- @return string
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
local function decl_words(atom)
local p = atom.paths or {}
return #(p.word_events or {})
end
local function count_kinds(decls)
local n = { atom = 0, atom_proc = 0, comp_bare = 0, comp_proc = 0 }
for _, a in ipairs(decls or {}) do
if n[a.kind] ~= nil then n[a.kind] = n[a.kind] + 1 end
end
return n
end
local function slot_suffix(key)
if type(key) ~= "string" or key:sub(1, 7) ~= "reguse:" then return nil end
return key:match("([^:]+)$")
end
local function decl_names(view)
local names = {}
for _, a in ipairs(view.decls or {}) do
if a.name then names[a.name] = true end
end
return names
end
local function path_in_module(path, view)
if type(path) ~= "string" or path == "" then return false end
local norm = path:gsub("\\", "/")
local dir = (view.dir or ""):gsub("\\", "/")
if dir ~= "" and (norm == dir or norm:sub(1, #dir + 1) == dir .. "/") then
return true
end
for _, src in ipairs(view.sources or {}) do
if (src.path or ""):gsub("\\", "/") == norm then return true end
end
return false
end
local function build_module_view(dir, dir_sources, corpus)
local decls = {}
for _, src in ipairs(dir_sources or {}) do
for _, a in ipairs((src.scan and src.scan.atoms) or {}) do
if not a.source_path then a.source_path = src.path end
decls[#decls + 1] = a
end
end
local dir_basename = source_basename(dir)
local sa = (corpus.static_analysis_results or {})[dir_basename] or {}
local schemas = {}
for name, schema in pairs(corpus.reg_use_schemas or {}) do
for _, a in ipairs(decls) do
if a.reg_use_schema_name == name then
schemas[#schemas + 1] = schema
break
end
end
end
return {
dir = dir,
sources = dir_sources or {},
decls = decls,
schemas = schemas,
findings = sa.findings or {},
sa = sa,
corpus = corpus,
}
end
local function render_section_declarations(add, view)
if #view.decls == 0 then add("_(none)_"); add(""); return end
add("| kind | name | source | line | words | min | max | branches | paths |")
add("|------|------|--------|------|-------|-----|-----|----------|-------|")
for _, a in ipairs(view.decls) do
local p = a.paths or {}
add(string.format("| %s | %s | %s | %d | %d | %s | %s | %s | %s |",
a.kind or "?",
a.name or "?",
source_basename(a.source_path or ""),
a.line or 0,
decl_words(a),
tostring(p.cycles_min or ""),
tostring(p.cycles_max or ""),
tostring(p.branches or ""),
tostring(p.paths or "")))
end
add("")
end
local function render_section_components(add, view)
local rows = {}
local index = (view.corpus and view.corpus.component_body_index) or {}
for _, a in ipairs(view.decls) do
if a.kind == "comp_bare" or a.kind == "comp_proc" then
local idx = index[a.name] or {}
local args = idx.arg_names or {}
rows[#rows + 1] = {
name = a.name,
kind = a.kind,
args = table.concat(args, ", "),
words = decl_words(a),
map = a.map_command or "",
}
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
add("| name | kind | arg_names | words | map |")
add("|------|------|-----------|-------|-----|")
for _, r in ipairs(rows) do
add(string.format("| %s | %s | %s | %d | %s |",
r.name, r.kind, r.args ~= "" and r.args or "", r.words, r.map))
end
add("")
end
local function render_section_reguse(add, view)
local wrote = false
for _, schema in ipairs(view.schemas or {}) do
wrote = true
add(string.format("### %s", schema.name or "?"))
for _, slot in ipairs(schema.slots or {}) do
local aliases = table.concat(slot.aliases or { slot.name }, ", ")
local ro = slot.readonly and " readonly" or ""
add(string.format("- slot `%s` aliases %s%s", slot.name, aliases, ro))
end
for _, a in ipairs(view.decls) do
if a.reg_use_schema_name == schema.name then
add(string.format("- bound `%s` param `%s`", a.name, a.reg_use_param_name or "?"))
end
end
add("")
end
local bound = {}
for _, schema in ipairs(view.schemas or {}) do
if schema.name then bound[schema.name] = true end
end
local errors = {}
for _, err in ipairs((view.corpus and view.corpus.reg_use_errors) or {}) do
if bound[err.schema_name] or path_in_module(err.source_file, view) then
errors[#errors + 1] = err
end
end
if #errors > 0 then
wrote = true
add("### parse errors")
for _, err in ipairs(errors) do
add(string.format("- `%s` %s", err.kind or "?", err.schema_name or ""))
end
add("")
end
if not wrote then add("_(none)_"); add("") end
end
local function render_section_annotations(add, view)
local rows = {}
for _, src in ipairs(view.sources) do
for _, info in ipairs((src.scan and src.scan.atom_infos) or {}) do
rows[#rows + 1] = {
source = source_basename(src.path),
line = info.info_line or 0,
name = info.atom_name or "?",
binds = info.binds or "",
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "",
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "",
phase = info.phase or "",
}
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
add("| source | line | name | binds | reads | writes | phase |")
add("|--------|------|------|-------|-------|--------|-------|")
for _, r in ipairs(rows) do
add(string.format("| %s | %d | %s | %s | %s | %s | %s |",
r.source, r.line, r.name, r.binds, r.reads, r.writes, r.phase))
end
add("")
end
local function render_section_component_annotations(add, view)
local rows = {}
for _, src in ipairs(view.sources) do
for _, info in ipairs((src.scan and src.scan.component_atom_infos) or {}) do
rows[#rows + 1] = {
source = source_basename(src.path),
line = info.info_line or 0,
name = info.atom_name or "?",
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "",
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "",
}
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
add("| source | line | name | reads | writes |")
add("|--------|------|------|-------|--------|")
for _, r in ipairs(rows) do
add(string.format("| %s | %d | %s | %s | %s |",
r.source, r.line, r.name, r.reads, r.writes))
end
add("")
end
local function render_section_binds(add, view)
local wrote = false
for _, src in ipairs(view.sources) do
for _, b in ipairs((src.scan and src.scan.binds) or {}) do
wrote = true
add(string.format("### %s (%s:%s, %s bytes)",
b.name, source_basename(src.path), tostring(b.line or 0), tostring(b.bytes or "")))
for _, f in ipairs(b.fields or {}) do
add(string.format("- `+%s %s`", tostring(f.offset or "?"), f.name or "?"))
end
add("")
end
end
if not wrote then add("_(none)_"); add("") end
end
local function render_section_phases(add, view)
local corpus = view.corpus or {}
local names = decl_names(view)
local wrote = false
for phase, entry in pairs(corpus.atom_phases or {}) do
local here = {}
for _, atom_name in ipairs(entry.atoms or {}) do
if names[atom_name] then here[#here + 1] = atom_name end
end
if #here > 0 then
wrote = true
add(string.format("- phase `%s`: %s", phase, table.concat(here, ", ")))
end
end
for name, entry in pairs(corpus.atom_views or {}) do
if names[name] then
wrote = true
add(string.format("- view `%s` binds `%s`", name, entry.binds_name or ""))
end
end
for name, entry in pairs(corpus.atom_ctxs or {}) do
if names[name] then
wrote = true
add(string.format("- ctx `%s` rbind `%s`", name, entry.rbind_atom or ""))
end
end
if not wrote then add("_(none)_") end
add("")
end
local function render_section_aliases(add, view)
local names = {}
local seen = {}
for _, src in ipairs(view.sources or {}) do
for name, entry in pairs((src.scan and src.scan.register_alias_registry) or {}) do
if not seen[name] then
seen[name] = entry
names[#names + 1] = name
end
end
end
table.sort(names)
if #names == 0 then add("_(none)_"); add(""); return end
add("| alias | type |")
add("|-------|------|")
for _, name in ipairs(names) do
local e = seen[name]
add(string.format("| %s | %s |", name, (e and e.default_type) or ""))
end
add("")
end
local function render_section_autoreg(add, view)
local allowed = decl_names(view)
for phase, entry in pairs((view.corpus and view.corpus.atom_phases) or {}) do
for _, atom_name in ipairs(entry.atoms or {}) do
if allowed[atom_name] then allowed[phase] = true end
end
end
local wrote = false
local seen = {}
local function dump(label, table_map)
local scopes = {}
for scope in pairs(table_map or {}) do
if allowed[scope] and not seen[label .. "\0" .. scope] then
scopes[#scopes + 1] = scope
end
end
table.sort(scopes)
for _, scope in ipairs(scopes) do
seen[label .. "\0" .. scope] = true
wrote = true
local syms = {}
for sym, gpr in pairs(table_map[scope] or {}) do
if type(gpr) == "string" and gpr ~= sym then
syms[#syms + 1] = string.format("%s → %s", sym, gpr)
else
syms[#syms + 1] = tostring(sym)
end
end
table.sort(syms)
add(string.format("- %s `%s`: %s", label, scope, table.concat(syms, ", ")))
end
end
local corpus = view.corpus or {}
dump("atom", corpus.atom_auto_regs)
dump("phase", corpus.phase_auto_regs)
for _, src in ipairs(view.sources or {}) do
dump("atom", src.scan and src.scan.atom_auto_regs)
dump("phase", src.scan and src.scan.phase_auto_regs)
end
if not wrote then add("_(none)_") end
add("")
end
local function render_section_collisions(add, view)
local rows = {}
for _, c in ipairs((view.corpus and view.corpus.collisions) or {}) do
local first = c.first_site or {}
local other = c.conflicting_site or {}
if path_in_module(first.path, view) or path_in_module(other.path, view) then
rows[#rows + 1] = c
end
end
if #rows == 0 then add("_(none)_"); add(""); return end
for _, c in ipairs(rows) do
local first = c.first_site or {}
local other = c.conflicting_site or {}
add(string.format("- `%s` `%s` first %s:%s conflict %s:%s",
c.kind or "?", c.name or "?",
tostring(first.path or "?"), tostring(first.line or "?"),
tostring(other.path or "?"), tostring(other.line or "?")))
end
add("")
end
local function render_section_findings(add, view)
local by_atom = {}
for _, f in ipairs(view.findings or {}) do
local key = f.atom or "?"
by_atom[key] = by_atom[key] or {}
by_atom[key][#by_atom[key] + 1] = f
end
if next(by_atom) == nil then add("_(none)_"); add(""); return end
local seen = {}
local function emit(name, fs)
add("### " .. name)
for _, f in ipairs(fs) do
local msg = f.msg or ""
local slot = slot_suffix(f.gpr_key or f.producer_destination)
if slot and not msg:find("(slot ", 1, true) then
msg = msg .. " (slot " .. slot .. ")"
end
add(string.format("- `[%s/%s] %s`", f.kind or "info", f.check or "?", msg))
end
add("")
end
for _, a in ipairs(view.decls) do
if by_atom[a.name] then
seen[a.name] = true
emit(a.name, by_atom[a.name])
end
end
local leftovers = {}
for name in pairs(by_atom) do
if not seen[name] then leftovers[#leftovers + 1] = name end
end
table.sort(leftovers)
for _, name in ipairs(leftovers) do emit(name, by_atom[name]) end
end
local function render_section_relations(add, view)
local wrote = false
for _, a in ipairs(view.decls) do
local rels = (a.paths and a.paths.relations) or {}
if #rels > 0 then
wrote = true
add("### " .. a.name)
for _, rel in ipairs(rels) do
local dest = rel.destination or rel.producer_destination or ""
local slot = slot_suffix(dest)
local dest_s = tostring(dest)
if slot then dest_s = dest_s .. " (slot " .. slot .. ")" end
add(string.format("- `%s` words %s → %s dest %s",
rel.semantic or "?",
tostring(rel.producer_word or "?"),
tostring(rel.consumer_word or "?"),
dest_s))
end
add("")
end
end
if not wrote then add("_(none)_"); add("") end
end
local HIDDEN_UNLESS_WRITTEN = {
R_AT = true, R_TapePtr = true, R_AtomJmp = true,
}
local PHYSICAL_GPR = {
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
R_T4 = true, R_T5 = true, R_T6 = true, R_T7 = true,
R_V0 = true, R_V1 = true,
}
local function encoder_wrote_key(atom, key)
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
for _, dest in pairs(ev.gpr_keys or {}) do
if dest == key then return true end
end
end
return false
end
local function written_name_for(key, atom)
local slot = key:match("^reguse:.+:(.+)$")
if slot then
local param = atom.reg_use_param_name
if param and param ~= "" then return param .. "." .. slot end
return slot
end
return key
end
local function aliases_for_key(key, atom, view)
local slot = key:match("^reguse:.+:(.+)$")
if not slot then return "" end
local schema_name = atom.reg_use_schema_name
local schema = view.corpus and view.corpus.reg_use_schemas and view.corpus.reg_use_schemas[schema_name]
if not schema then return "" end
for _, s in ipairs(schema.slots or {}) do
if s.name == slot then
local names = {}
for _, alias in ipairs(s.aliases or {}) do
if alias ~= slot then names[#names + 1] = alias end
end
if #names == 0 then
if s.aliases and #s.aliases > 0 then return table.concat(s.aliases, ", ") end
return ""
end
return table.concat(names, ", ")
end
end
return ""
end
local function physical_for_key(key, atom, view)
if PHYSICAL_GPR[key] then return key end
local corpus = view.corpus or {}
local alias = (corpus.register_alias_registry or {})[key]
if type(alias) == "table" then
local phys = alias.physical or alias.gpr or alias.code_name
if type(phys) == "string" and PHYSICAL_GPR[phys] then return phys end
if type(alias.name) == "string" and PHYSICAL_GPR[alias.name] then return alias.name end
elseif type(alias) == "string" and PHYSICAL_GPR[alias] then
return alias
end
local atom_map = (corpus.atom_auto_regs or {})[atom.name]
if type(atom_map) == "table" then
local slot = key:match("^reguse:.+:(.+)$") or key
local bound = atom_map[slot] or atom_map["R_" .. slot]
if type(bound) == "string" and PHYSICAL_GPR[bound] then return bound end
end
return ""
end
local function last_relation_for(key, atom)
local last = nil
for _, rel in ipairs((atom.paths and atom.paths.relations) or {}) do
local dest = rel.destination or rel.producer_destination
if dest == key then last = rel end
end
if not last then return "" end
local sem = last.semantic or "?"
local a = last.producer_word
local b = last.consumer_word
if a and b then return string.format("%s w%s→%s", sem, tostring(a), tostring(b)) end
return sem
end
local function render_section_forward(add, view)
local wrote = false
for _, a in ipairs(view.decls) do
local gpr = a.paths and a.paths.forward_state and a.paths.forward_state.gpr_values
local keys = {}
for k in pairs(gpr or {}) do
if k == "R_0" then
-- hidden
elseif HIDDEN_UNLESS_WRITTEN[k] and not encoder_wrote_key(a, k) then
-- hidden
else
keys[#keys + 1] = k
end
end
if #keys > 0 then
wrote = true
add("### " .. a.name)
add("| written | aliases | physical | lattice | last relation |")
add("|---|---|---|---|---|")
table.sort(keys)
for _, k in ipairs(keys) do
local slot = gpr[k]
local lattice = ""
if slot and slot.kind == "constant" then
lattice = tostring(slot.value)
end
add(string.format("| `%s` | %s | %s | %s | %s |",
written_name_for(k, a),
aliases_for_key(k, a, view),
physical_for_key(k, a, view),
lattice,
last_relation_for(k, a)))
end
add("")
end
end
if not wrote then add("_(none)_"); add("") end
end
local SECTION_RENDERERS = {
{ header = "## Declarations", render = render_section_declarations },
{ header = "## Components", render = render_section_components },
{ header = "## RegUse schemas", render = render_section_reguse },
{ header = "## Annotations", render = render_section_annotations },
{ header = "## Component annotations", render = render_section_component_annotations },
{ header = "## Binds_* structs", render = render_section_binds },
{ header = "## Phases / views / ctx", render = render_section_phases },
{ header = "## Register aliases", render = render_section_aliases },
{ header = "## Auto-reg", render = render_section_autoreg },
{ header = "## Collisions", render = render_section_collisions },
{ header = "## Findings", render = render_section_findings },
{ header = "## Relations", render = render_section_relations },
{ header = "## GPR model", render = render_section_forward },
}
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
--- One ModuleView from the corpus; SECTION_RENDERERS walks it.
--- @param view table
--- @return string
local function render_module_meta_report(view)
local dir_basename = source_basename(view.dir)
local lines = {
"# " .. dir_basename .. " — atom meta report",
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
@@ -264,199 +786,41 @@ local function render_module_meta_report(dir, dir_sources, annot_results, sa_res
}
local function add(s) lines[#lines + 1] = s end
-- Module summary table.
local n_atoms = 0
local n_annot = 0
local n_binds = 0
local n_macros = 0
local n_bare, n_proc = 0, 0
for _, r in ipairs(annot_results) do
n_atoms = n_atoms + #r.atoms
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
local kinds = count_kinds(view.decls)
local n_annot, n_binds, n_macros = 0, 0, 0
for _, src in ipairs(view.sources) do
n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
n_binds = n_binds + #((src.scan and src.scan.binds) or {})
n_macros = n_macros + #((src.scan and src.scan.macros) or {})
end
for _, a in ipairs(sa_results.atoms or {}) do
if a.kind == "comp_bare" then n_bare = n_bare + 1
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
local n_err, n_warn, n_info = 0, 0, 0
for _, f in ipairs(view.findings or {}) do
if f.kind == "error" then n_err = n_err + 1
elseif f.kind == "warning" then n_warn = n_warn + 1
else n_info = n_info + 1
end
end
add("## Module summary"); add("")
add("| metric | value |"); add("|--------|-------|")
add(string.format("| sources | %d |", #dir_sources))
add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
#(sa_results.atoms or {}),
#(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
add(string.format("| sources | %d |", #view.sources))
add(string.format("| decls | %d (atom: %d, atom_proc: %d, comp_bare: %d, comp_proc: %d) |",
#view.decls, kinds.atom, kinds.atom_proc, kinds.comp_bare, kinds.comp_proc))
add(string.format("| annotations | %d |", n_annot))
add(string.format("| binds structs | %d |", n_binds))
add(string.format("| macro decls | %d |", n_macros))
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
#(sa_results.findings or {}),
#(sa_results.errors or {}),
#(sa_results.warnings or {}),
#(sa_results.info or {})))
#(view.findings or {}), n_err, n_warn, n_info))
add("")
-- Sources
add("## Sources"); add("")
for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
for _, s in ipairs(view.sources) do add("- `" .. s.path .. "`") end
add("")
-- Atoms (annotation)
add("## Atoms"); add("")
add("| kind | name | source | line |"); add("|------|------|--------|------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.atoms) do
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
end
for _, row in ipairs(SECTION_RENDERERS) do
add(row.header); add("")
row.render(add, view)
end
add("")
-- Annotations
add("## Annotations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | name | binds | reads | writes |")
add("|--------|------|------|-------|-------|--------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, a in ipairs(r.annots) do
local binds = a.binds or ""
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or ""
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or ""
add(string.format("| %s | %d | %s | %s | %s | %s |"
, src_name, a.line, a.name, binds, reads, writes))
end
end
end
add("")
-- Binds_* structs
add("## Binds_* structs"); add("")
if #annot_results == 0 then
add("_(none)_")
else
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, b in ipairs(r.binds) do
add(string.format("### %s (%s:%d, %d bytes)",
b.name, src_name, b.line, b.bytes))
for _, f in ipairs(b.fields) do
add(string.format("- `+%d %s`", f.offset, f.name))
end
add("")
end
end
end
-- Macro decls
add("## Macro word-count declarations"); add("")
if #annot_results == 0 then
add("_(none)_")
else
add("| source | line | macro declaration |")
add("|--------|------|-------------------|")
for _, r in ipairs(annot_results) do
local src_name = source_basename(r.source)
for _, m in ipairs(r.macros) do
add(string.format("| %s | %d | %s |",
src_name, m.line, m.name))
end
end
end
add("")
-- Findings by atom (static-analysis)
add("## Static analysis — findings by atom"); add("")
local by_atom = {}
for _, f in ipairs(sa_results.findings or {}) do
by_atom[f.atom] = by_atom[f.atom] or {}
by_atom[f.atom][#by_atom[f.atom] + 1] = f
end
if next(by_atom) == nil then
add("_(no findings)_")
else
for _, a in ipairs(sa_results.atoms or {}) do
local fs = by_atom[a.name]
if fs then
add(string.format("### %s", a.name))
for _, f in ipairs(fs) do
add(string.format("- `[%s] %s`", f.check, f.msg))
end
add("")
end
end
end
-- Errors / Warnings / Info
local function add_findings(label, entries)
add(string.format("## %s", label))
if #entries == 0 then
add("_(none)_")
else
for _, e in ipairs(entries) do
add(string.format("- line %d %s", e.line, e.msg))
end
end
add("")
end
add_findings("Errors", sa_results.errors or {})
add_findings("Warnings", sa_results.warnings or {})
add_findings("Info", sa_results.info or {})
-- Per-atom cycle counts (path-aware)
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
add("| atom | source | min | max | branches | paths | notes |")
add("|------|--------|-----|-----|----------|-------|-------|")
local sorted = {}
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
table.sort(sorted, function(x, y)
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
end)
for _, a in ipairs(sorted) do
local p = a.paths or {}
local src_name = a.source_path and source_basename(a.source_path) or ""
local notes = ""
if p.has_loops then notes = notes .. " [loop!]" end
if p.unknown_macros and #p.unknown_macros > 0 then
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
end
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
a.name, src_name,
p.cycles_min or 0, p.cycles_max or 0,
p.branches or 0, p.paths or 0, notes))
end
add("")
-- Per-source scan summary
add("## Per-source scan summary"); add("")
for _, src in ipairs(dir_sources) do
local src_atoms = {}
for _, a in ipairs(sa_results.atoms or {}) do
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
end
if #src_atoms > 0 then
local mn, mx = math.huge, -1
for _, a in ipairs(src_atoms) do
local p = a.paths or {}
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
end
local path_str
if mx > 0 then
path_str = string.format(" cycles=%d..%d", mn, mx)
else
path_str = string.format(" %d cycles", mn)
end
add(string.format("- `%s` — %d atom%s%s",
src.basename, #src_atoms,
#src_atoms == 1 and "" or "s", path_str))
end
end
add("")
return table.concat(lines, "\n") .. "\n"
end
@@ -474,19 +838,8 @@ local REPORT_RENDERERS = {
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
once = false,
gather = function(ctx, dir, dir_sources)
-- Annotations: re-run `annotation.validate()` per source (the existing pattern).
local annot_results = {}
for _, src in ipairs(dir_sources) do
if src.scan then
local r = annotation.validate(ctx, src, nil)
r.source = src.path
annot_results[#annot_results + 1] = r
end
end
-- Static-analysis: read stashed projection (no re-validate).
local dir_basename = dir:match("([^/\\]+)$") or dir
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
local corpus = ctx.shared.corpus
return render_module_meta_report(build_module_view(dir, dir_sources, corpus))
end,
},
{
@@ -554,32 +907,30 @@ function M.run(ctx)
end
end
-- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
local annot_results = {}
local view = build_module_view(dir, dir_sources, corpus)
local n_annot, n_binds, n_macros = 0, 0, 0
for _, src in ipairs(dir_sources) do
if src.scan then
local r = annotation.validate(ctx, src, nil)
r.source = src.path
annot_results[#annot_results + 1] = r
n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
n_binds = n_binds + #((src.scan and src.scan.binds) or {})
n_macros = n_macros + #((src.scan and src.scan.macros) or {})
end
local n_err, n_warn, n_info = 0, 0, 0
for _, f in ipairs(view.findings or {}) do
if f.kind == "error" then n_err = n_err + 1
elseif f.kind == "warning" then n_warn = n_warn + 1
else n_info = n_info + 1
end
end
local n_annot, n_binds, n_macros = 0, 0, 0
for _, r in ipairs(annot_results) do
n_annot = n_annot + #r.annots
n_binds = n_binds + #r.binds
n_macros = n_macros + #r.macros
end
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
all_modules[#all_modules + 1] = {
module = dir_basename,
atoms = #(sa_results.atoms or {}),
atoms = #view.decls,
annots = n_annot,
binds = n_binds,
macros = n_macros,
findings = #(sa_results.findings or {}),
errors = #(sa_results.errors or {}),
warnings = #(sa_results.warnings or {}),
info = #(sa_results.info or {}),
findings = #(view.findings or {}),
errors = n_err,
warnings = n_warn,
info = n_info,
}
end
+616 -85
View File
@@ -6,6 +6,7 @@
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
--- MipsAtomComp_ (kind = "comp_bare")
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
--- MipsAtomComp_ProcMap_ (kind = "comp_proc", body is the one command)
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
--- MipsCode code_<name> (kind = "raw_atom", offsets pass only)
--- typedef Struct_(Binds_X) { fields }
@@ -415,31 +416,51 @@ local function walk_body_fields(body, build_field)
return fields
end
-- Parse the `<type> <field>;` declarations from a Struct_ body.
-- Parse the `<type> <field>[, <field>...];` declarations from a Struct_ body.
-- After the type and `*` chain, keep reading `, ident` until `;`.
-- Same type, same pointer depth for every name on that list.
-- Returns the raw fields array with `{name, type_name, pointer_depth}` only (NO offset / byte_size).
-- The propagation pass `resolve_struct_field_sizes` walks each struct's fields AFTER type resolution and populates offset + byte_size in place.
-- Returns (fields). The aggregate byte_count is computed in the propagation pass (it depends on whether every field's type resolved).
local function parse_struct_body_fields(body)
return walk_body_fields(body, function(type_name, type_end, after_type)
-- Parse the trailing `*` chain to derive pointer_depth.
local depth, cursor = 0, after_type
while cursor <= #body and body:sub(cursor, cursor) == "*" do
depth = depth + 1
cursor = cursor + 1
cursor = duffle.skip_ws_and_cmt(body, cursor)
local fields = {}
local body_pos = 1
local body_len = #body
while body_pos <= body_len do
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
if body_pos > body_len then break end
local type_name, type_end = duffle.read_ident(body, body_pos)
if not type_name then
body_pos = body_pos + 1
else
local depth, cursor = 0, duffle.skip_ws_and_cmt(body, type_end)
while cursor <= body_len and body:sub(cursor, cursor) == "*" do
depth = depth + 1
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
end
while cursor <= body_len do
local field_ident, field_end = duffle.read_ident(body, cursor)
if not field_ident then break end
fields[#fields + 1] = {
name = field_ident,
type_name = type_name,
pointer_depth = depth,
offset = nil,
byte_size = nil,
}
cursor = duffle.skip_ws_and_cmt(body, field_end)
if body:sub(cursor, cursor) == "," then
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
else
break
end
end
if cursor <= body_len and body:sub(cursor, cursor) == ";" then
cursor = cursor + 1
end
body_pos = cursor
end
-- Read the field ident immediately after the type chain.
local field_ident, field_end = duffle.read_ident(body, cursor)
if not field_ident then return nil, type_end + 1 end
return {
name = field_ident,
type_name = type_name,
pointer_depth = depth,
-- offset + byte_size filled by resolve_struct_field_sizes
offset = nil,
byte_size = nil,
}, field_end
end)
end
return fields
end
-- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries.
@@ -1251,6 +1272,53 @@ local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
return after_paren
end
--- Lookahead for `atom_info(...)` after a declaration's closing paren.
--- Records into `dest` (atom_infos or component_atom_infos). Returns the position after the info, or after_paren if none.
local function parse_atom_info_after_decl(source, after_paren, raw_name, line_of, out, dest)
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
local look_ident, look_end = duffle.read_ident(source, lookahead)
if look_ident ~= "atom_info" then return after_paren end
local info_open = duffle.skip_ws_and_cmt(source, look_end)
if source:sub(info_open, info_open) ~= "(" then return after_paren end
local info_inner, info_after = duffle.read_parens(source, info_open)
if not info_inner then return after_paren end
local info_line = line_of(info_open)
local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line)
dest = dest or out.atom_infos
dest[#dest + 1] = {
atom_name = raw_name or "?", binds = ai_binds,
reads = ai_reads or {}, writes = ai_writes or {},
view = ai_view,
reg_type_overrides = ai_overrides,
ctx_atom = ai_ctx,
phase = ai_phase,
info_line = line_of(lookahead),
}
if ai_view and raw_name then
out.atom_views[raw_name] = {
atom_name = raw_name,
binds_name = ai_view,
reg_type_overrides = ai_overrides,
info_line = line_of(lookahead),
}
elseif raw_name and ai_overrides then
out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) }
out.atom_views[raw_name].reg_type_overrides = ai_overrides
end
if raw_name then
if ai_ctx then
out.atom_ctxs = out.atom_ctxs or {}
out.atom_ctxs[raw_name] = { rbind_atom = ai_ctx, info_line = line_of(lookahead), source = source }
end
if ai_phase then
out.atom_phases = out.atom_phases or {}
out.atom_phases[ai_phase] = out.atom_phases[ai_phase] or { atoms = {} }
out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name
end
end
return info_after
end
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
--- @param source string
--- @param pos integer
@@ -1264,53 +1332,8 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
local raw_name = duffle.read_ident(inner, 1)
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
local brace_search_pos = after_paren
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
local look_ident, look_end = duffle.read_ident(source, lookahead)
if look_ident == "atom_info" then
local info_open = duffle.skip_ws_and_cmt(source, look_end)
if source:sub(info_open, info_open) == "(" then
local info_inner, info_after = duffle.read_parens(source, info_open)
-- info_line feeds the per-atom reg_type_overrides table.
local info_line = line_of(info_open)
local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line)
out.atom_infos[#out.atom_infos + 1] = {
atom_name = raw_name or "?", binds = ai_binds,
reads = ai_reads or {}, writes = ai_writes or {},
view = ai_view,
reg_type_overrides = ai_overrides,
ctx_atom = ai_ctx,
phase = ai_phase,
info_line = line_of(lookahead),
}
if ai_view and raw_name then
out.atom_views[raw_name] = {
atom_name = raw_name,
binds_name = ai_view,
reg_type_overrides = ai_overrides,
info_line = line_of(lookahead),
}
elseif raw_name and ai_overrides then
-- Record per-atom overrides even without atom_view.
out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) }
out.atom_views[raw_name].reg_type_overrides = ai_overrides
end
-- Project the per-atom atom_ctx / atom_phase declarations onto the global phase index.
if raw_name then
if ai_ctx then
out.atom_ctxs = out.atom_ctxs or {}
out.atom_ctxs[raw_name] = { rbind_atom = ai_ctx, info_line = line_of(lookahead), source = source }
end
if ai_phase then
out.atom_phases = out.atom_phases or {}
out.atom_phases[ai_phase] = out.atom_phases[ai_phase] or { atoms = {} }
out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name
end
end
brace_search_pos = info_after
end
end
-- Lookahead for atom_info(...) between `)` and `{`.
local brace_search_pos = parse_atom_info_after_decl(source, after_paren, raw_name, line_of, out, out.atom_infos)
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
if not body then return after_brace end
@@ -1335,7 +1358,9 @@ local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
local raw_name = duffle.read_ident(inner, 1)
if not raw_name then return open_paren + 1 end
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
out.component_atom_infos = out.component_atom_infos or {}
local brace_search_pos = parse_atom_info_after_decl(source, after_paren, strip_ac_prefix(raw_name), line_of, out, out.component_atom_infos)
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
if not body then return after_brace end
local name = strip_ac_prefix(raw_name)
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
@@ -1380,15 +1405,35 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
return after_paren
end
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
--- Parse: `MipsAtomComp_ProcMap_(ab, command)` — body is the one command (second arg).
--- Reuses the proc name walk. Kind is `comp_proc`. The C expansion wraps
--- `atom_dbg_skip MipsAtomComp_Proc_(ab, {command })`; source-as-written is the map.
--- @param source string
--- @param pos integer
--- @param ident_end integer
--- @param line_of fun(pos: integer): integer
--- @param out SourceScan
--- @return integer
local function parse_mips_atom_comp_proc_map(source, pos, ident_end, line_of, out)
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
if not inner then return after_paren end
local args = duffle.split_top_level_commas(inner)
if #args < 2 then return after_paren end
local command = duffle.trim(args[2])
if command == "" then return after_paren end
local raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
if not raw_name then raw_name = "?" end
local name = strip_ac_prefix(raw_name)
local body_off = open_paren + 1 + (inner:find(command, 1, true) or 1) - 1
register_atom(out, "comp_proc", line_of(pos), name, command, body_off, raw_name, pos, after_paren, source)
local entry = out.atoms[#out.atoms]
entry.map_command = command
return after_paren
end
--- Parse: `MipsAtom_Proc_(aa, { body })` — body is inside the LAST `{` in args.
--- Kind is `atom_proc`. The name is the preceding function ident as written.
--- Offsets walk this kind. Components do not emit a `mac_*` alias for it.
--- @param source string
--- @param pos integer
--- @param ident_end integer
@@ -1412,17 +1457,57 @@ local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
if close_pos > #inner + 1 then return after_paren end
-- The atom name is derived from the preceding function declaration
-- (`internal MipsAtom* X_proc(...)`), not from the first macro arg (which
-- is now `aa`). The backward walk finds the function decl before open_paren
-- and strips the `_proc` suffix.
local raw_name = duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
-- The atom name is the preceding function ident as written
-- (`internal MipsAtom* X(...)`). The first macro arg is the arena.
local raw_name, args_inner, func_ident, after_func_paren =
duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
if not raw_name then raw_name = "?" end
local name = strip_ac_prefix(raw_name)
local name = strip_ac_prefix(raw_name)
if after_func_paren then
parse_atom_info_after_decl(source, after_func_paren, name, line_of, out, out.atom_infos)
else
parse_atom_info_after_decl(source, pos, name, line_of, out, out.atom_infos)
end
local reg_use_schema_name = nil
local reg_use_param_name = nil
if args_inner then
local arg_tokens = duffle.split_top_level_commas(args_inner)
for _, tok in ipairs(arg_tokens) do
local trimmed = duffle.trim(tok)
local schema_suffix, param = trimmed:match("RegUse_([%w_]+)%s+([%w_]+)$")
if schema_suffix then
if reg_use_schema_name then
out.reg_use_errors[#out.reg_use_errors + 1] = {
kind = "reguse_multiple_params",
schema_name = "RegUse_" .. schema_suffix,
source_line = line_of(pos),
}
else
reg_use_schema_name = "RegUse_" .. schema_suffix
reg_use_param_name = param
end
end
end
end
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
local body_off = open_paren + 2 + last_brace_pos
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
local entry = out.atoms[#out.atoms]
entry.reg_use_schema_name = reg_use_schema_name
entry.reg_use_param_name = reg_use_param_name
if reg_use_schema_name and func_ident then
local expected = "RegUse_" .. func_ident
if reg_use_schema_name ~= expected then
out.reg_use_errors[#out.reg_use_errors + 1] = {
kind = "reguse_name_mismatch",
schema_name = reg_use_schema_name,
func_ident = func_ident,
source_line = line_of(pos),
}
end
end
return after_paren
end
@@ -1530,6 +1615,236 @@ local function register_typedef_alias(underlying, name, pos, line_of, out)
}
end
local parse_reg_use_schema_body
local function fields_for_reg_type(type_name, type_registry)
local reg_name = "Reg_" .. type_name
local entry = type_registry and type_registry[reg_name]
if entry and entry.fields and #entry.fields > 0 then
local names = {}
for _, field in ipairs(entry.fields) do
if field.name then names[#names + 1] = field.name end
end
if #names > 0 then return names end
end
if entry and entry.body and parse_reg_use_schema_body then
local schema = parse_reg_use_schema_body(entry.body, type_registry)
if schema and schema.slots then
local names = {}
for _, slot in ipairs(schema.slots) do
if slot.name then names[#names + 1] = slot.name end
end
if #names > 0 then return names end
end
end
return nil
end
parse_reg_use_schema_body = function(body, type_registry, opts)
opts = opts or {}
local require_types = opts.require_types == true
local pending = false
local slots = {}
local alias_to_slot = {}
local slot_names = {}
local errors = {}
local function add_alias(path, slot)
if alias_to_slot[path] then
errors[#errors + 1] = { kind = "reguse_duplicate_alias", path = path }
return false
end
alias_to_slot[path] = slot
return true
end
local function add_slot(name, aliases, readonly)
if slot_names[name] then
errors[#errors + 1] = { kind = "reguse_duplicate_slot", name = name }
return nil
end
slot_names[name] = true
local slot = { name = name, aliases = aliases, readonly = readonly == true }
slots[#slots + 1] = slot
return slot
end
local function parse_reg_names(text, pos)
local names = {}
while pos <= #text do
pos = duffle.skip_ws_and_cmt(text, pos)
local name, name_end = duffle.read_ident(text, pos)
if not name then return nil, pos end
names[#names + 1] = name
pos = duffle.skip_ws_and_cmt(text, name_end)
if text:sub(pos, pos) == "," then
pos = pos + 1
else
break
end
end
if text:sub(pos, pos) == ";" then pos = pos + 1 end
return names, pos
end
local pos = 1
while pos <= #body do
pos = duffle.skip_ws_and_cmt(body, pos)
if pos > #body then break end
local first, first_end = duffle.read_ident(body, pos)
if not first then
pos = pos + 1
goto continue
end
local after = duffle.skip_ws_and_cmt(body, first_end)
if first == "const" then
errors[#errors + 1] = { kind = "reguse_const_reg_spelling" }
return nil, errors
elseif first == "union" then
if body:sub(after, after) ~= "{" then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local inner, after_braces = duffle.read_braces(body, after)
if not inner then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local members = {}
local union_readonly = nil
local inner_pos = 1
while inner_pos <= #inner do
inner_pos = duffle.skip_ws_and_cmt(inner, inner_pos)
if inner_pos > #inner then break end
local m_type, m_type_end = duffle.read_ident(inner, inner_pos)
if not m_type then
inner_pos = inner_pos + 1
goto continue_inner
end
if m_type == "const" then
errors[#errors + 1] = { kind = "reguse_const_reg_spelling" }
return nil, errors
end
if m_type ~= "Reg" then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local m_after = duffle.skip_ws_and_cmt(inner, m_type_end)
local m_readonly = false
local maybe_const, maybe_end = duffle.read_ident(inner, m_after)
if maybe_const == "const" then
m_readonly = true
m_after = duffle.skip_ws_and_cmt(inner, maybe_end)
end
if union_readonly == nil then
union_readonly = m_readonly
elseif union_readonly ~= m_readonly then
errors[#errors + 1] = { kind = "reguse_mixed_const" }
return nil, errors
end
local names, new_inner = parse_reg_names(inner, m_after)
if not names or #names == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
for _, n in ipairs(names) do members[#members + 1] = n end
inner_pos = new_inner
::continue_inner::
end
if #members == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local after_close = duffle.skip_ws_and_cmt(body, after_braces)
local inst_name, inst_end = duffle.read_ident(body, after_close)
local aliases = {}
local slot_name
if inst_name then
slot_name = inst_name
for _, m in ipairs(members) do
local path = inst_name .. "." .. m
if not add_alias(path, slot_name) then return nil, errors end
aliases[#aliases + 1] = path
end
after_close = inst_end
else
slot_name = members[1]
for _, m in ipairs(members) do
if not add_alias(m, slot_name) then return nil, errors end
aliases[#aliases + 1] = m
end
end
if not add_slot(slot_name, aliases, union_readonly) then return nil, errors end
after_close = duffle.skip_ws_and_cmt(body, after_close)
if body:sub(after_close, after_close) == ";" then after_close = after_close + 1 end
pos = after_close
elseif first == "Reg" or first == "Reg_" then
local typed_fields = nil
if first == "Reg_" then
if body:sub(after, after) ~= "(" then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local type_inner, after_paren = duffle.read_parens(body, after)
if not type_inner then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
local type_ident = duffle.trim(type_inner)
typed_fields = fields_for_reg_type(type_ident, type_registry)
if not typed_fields then
if require_types then
errors[#errors + 1] = { kind = "reguse_unknown_reg_type", type_name = type_ident }
else
pending = true
end
end
after = duffle.skip_ws_and_cmt(body, after_paren)
end
local readonly = false
local maybe_const, maybe_end = duffle.read_ident(body, after)
if maybe_const == "const" then
readonly = true
after = duffle.skip_ws_and_cmt(body, maybe_end)
end
local names, new_pos = parse_reg_names(body, after)
if not names or #names == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
for _, n in ipairs(names) do
if first == "Reg_" then
if typed_fields then
for _, field in ipairs(typed_fields) do
local path = n .. "." .. field
if not add_alias(path, path) then return nil, errors end
if not add_slot(path, { path }, readonly) then return nil, errors end
end
end
else
if not add_alias(n, n) then return nil, errors end
if not add_slot(n, { n }, readonly) then return nil, errors end
end
end
pos = new_pos
else
errors[#errors + 1] = { kind = "reguse_malformed" }
return nil, errors
end
::continue::
end
if #slots == 0 then
if pending and not require_types then
return { slots = slots, alias_to_slot = alias_to_slot, pending = true }, errors
end
if #errors == 0 then
errors[#errors + 1] = { kind = "reguse_malformed" }
end
return nil, errors
end
return { slots = slots, alias_to_slot = alias_to_slot, pending = pending }, errors
end
--- Parse: `typedef` declarations.
---
--- Recognizes four shapes:
@@ -1563,6 +1878,21 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
if not body then return after_brace end
register_struct_type(body, name, pos, line_of, out)
if name:sub(1, 7) == "RegUse_" then
local schema, schema_errors = parse_reg_use_schema_body(body, out.type_name_registry)
if schema then
schema.name = name
schema.source_file = out._source_file
schema.source_line = line_of(pos)
out.reg_use_schemas[name] = schema
end
for _, err in ipairs(schema_errors or {}) do
err.schema_name = name
err.source_file = out._source_file
err.source_line = line_of(pos)
out.reg_use_errors[#out.reg_use_errors + 1] = err
end
end
attach_debug_skip_marker(out, "unrelated")
return after_brace
@@ -1893,6 +2223,7 @@ local DECL_PARSERS = {
MipsAtom_Proc_ = parse_mips_atom_proc,
MipsAtomComp_ = parse_mips_atom_comp,
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
MipsAtomComp_ProcMap_ = parse_mips_atom_comp_proc_map,
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
-- identifier follows the ordinary unrelated-token path; there is no alias.
atom_dbg_skip = parse_dbg_skip_marker,
@@ -1912,6 +2243,137 @@ local DECL_PARSERS = {
-- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`.
-- Unknown identifiers follow the same unrelated-token path as every other unsupported source token.
local TAPE_SKIP_MACROS = {
MipsAtom_ = true,
MipsAtom_Proc_ = true,
MipsAtomComp_ = true,
MipsAtomComp_Proc_ = true,
MipsAtomComp_ProcMap_ = true,
Struct_ = true,
Enum_ = true,
}
local function collect_addrs_assigns(text)
local addrs = {}
local pos = 1
local n = #text
while pos <= n do
pos = duffle.skip_ws_and_cmt(text, pos)
if pos > n then break end
local ident, ident_end = duffle.read_ident(text, pos)
if ident == "addrs" then
local after = duffle.skip_ws_and_cmt(text, ident_end)
if text:sub(after, after) == "[" then
local inner, after_br = duffle.read_brackets(text, after)
local idx = inner and tonumber(duffle.trim(inner))
after_br = duffle.skip_ws_and_cmt(text, after_br or after)
if idx and text:sub(after_br, after_br) == "=" then
local rhs = duffle.skip_ws_and_cmt(text, after_br + 1)
local rhs_ident = duffle.read_ident(text, rhs)
if rhs_ident then addrs[idx] = rhs_ident end
pos = rhs
else
pos = after_br or (after + 1)
end
else
pos = ident_end
end
elseif ident then
pos = ident_end
else
pos = pos + 1
end
end
return addrs
end
local function collect_tb_emits(body, addrs)
local names = {}
local pos = 1
local n = #body
while pos <= n do
pos = duffle.skip_ws_and_cmt(body, pos)
if pos > n then break end
local ident, ident_end = duffle.read_ident(body, pos)
if ident == "tb_emit_" or ident == "tb_emit" then
local after = duffle.skip_ws_and_cmt(body, ident_end)
if body:sub(after, after) == "(" then
local inner, after_p = duffle.read_parens(body, after)
local name
if ident == "tb_emit_" then
name = duffle.trim(inner or ""):match("^([%w_]+)")
else
local args = duffle.split_top_level_commas(inner or "")
local last = duffle.trim(args[#args] or "")
local idx = last:match("^addrs%s*%[%s*(%d+)%s*%]$")
if idx then
name = addrs[tonumber(idx)]
else
name = last:match("([%w_]+)$")
end
end
if name then names[#names + 1] = name end
pos = after_p or (after + 1)
else
pos = ident_end
end
elseif ident then
pos = ident_end
else
pos = pos + 1
end
end
return names
end
-- Linear appearance order of tb_emit / tb_emit_ in each C function body.
-- Commented-out emits are skipped by skip_ws_and_cmt. No C if/loop CFG.
local function scan_tape_chains(source)
local addrs = collect_addrs_assigns(source)
local chains = {}
local pos = 1
local n = #source
while pos <= n do
pos = duffle.skip_ws_and_cmt(source, pos)
if pos > n then break end
local ident, ident_end = duffle.read_ident(source, pos)
if ident and TAPE_SKIP_MACROS[ident] then
local after = duffle.skip_ws_and_cmt(source, ident_end)
if source:sub(after, after) == "(" then
local _, after_p = duffle.read_parens(source, after)
after = duffle.skip_ws_and_cmt(source, after_p or after)
end
if source:sub(after, after) == "{" then
local _, after_b = duffle.read_braces(source, after)
pos = after_b or (after + 1)
else
pos = after
end
elseif ident then
local after = duffle.skip_ws_and_cmt(source, ident_end)
if source:sub(after, after) == "(" then
local _, after_p = duffle.read_parens(source, after)
after = duffle.skip_ws_and_cmt(source, after_p or after)
if source:sub(after, after) == "{" then
local body, after_b = duffle.read_braces(source, after)
local names = collect_tb_emits(body or "", addrs)
if #names > 0 then
chains[#chains + 1] = names
end
pos = after_b or (after + 1)
else
pos = after
end
else
pos = ident_end
end
else
pos = pos + 1
end
end
return chains
end
-- ════════════════════════════════════════════════════════════════════════════
-- The single source walker
-- ════════════════════════════════════════════════════════════════════════════
@@ -1930,6 +2392,7 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
raw_atoms = {},
binds = {},
atom_infos = {},
component_atom_infos = {},
macros = {},
-- Raw marker evidence for annotation validation. The `debug_skip` boolean
-- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip.
@@ -1954,6 +2417,8 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
-- typedef chain walking (cycle-guarded, depth <= 8), and struct field sums.
-- See `propagate_type_sizes()` below.
type_name_registry = {},
reg_use_schemas = {},
reg_use_errors = {},
-- Shared `R_*_Code -> integer code` registry
-- (passed in from M.run pass 1; same reference so preprocessor intercept writes are visible to the enum-value resolver).
-- Stripped from `src.scan` before return.
@@ -2021,6 +2486,7 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
-- Runs AFTER the source walk so all typedef / Struct_ / Enum_ declarations have been parsed into `out.type_name_registry`.
-- Mutates each entry's `byte_size` field in place; fields with pointer_depth > 0 already carry byte_size = 4 from parse time and are unaffected.
propagate_type_sizes(out)
out.tape_chains = scan_tape_chains(source)
return out
end
@@ -2183,16 +2649,21 @@ local function merge_corpus_registries(corpus)
corpus.atom_ctxs = corpus.atom_ctxs or {}
corpus.atom_phases = corpus.atom_phases or {}
corpus.atom_infos = corpus.atom_infos or {}
corpus.component_atom_infos = corpus.component_atom_infos or {}
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
corpus.collisions = corpus.collisions or {}
corpus.reg_use_schemas = corpus.reg_use_schemas or {}
corpus.reg_use_errors = corpus.reg_use_errors or {}
corpus.tape_chains = corpus.tape_chains or {}
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
-- This is safe because M.run is the only writer to these tables within a single orchestrator invocation.
for _, key in ipairs({
"register_alias_registry", "type_name_registry", "binds_by_name",
"atoms_by_name", "atom_views", "atom_ctxs", "atom_phases",
"atom_infos", "collisions",
"atom_infos", "component_atom_infos", "collisions", "reg_use_schemas", "reg_use_errors",
"tape_chains",
}) do
corpus[key] = {}
end
@@ -2285,6 +2756,65 @@ local function merge_corpus_registries(corpus)
for _, info in ipairs(scan.atom_infos or {}) do
corpus.atom_infos[#corpus.atom_infos + 1] = info
end
for _, info in ipairs(scan.component_atom_infos or {}) do
corpus.component_atom_infos[#corpus.component_atom_infos + 1] = info
end
for name, schema in pairs(scan.reg_use_schemas or {}) do
if corpus.reg_use_schemas[name] == nil then
corpus.reg_use_schemas[name] = schema
end
end
for _, err in ipairs(scan.reg_use_errors or {}) do
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
end
for _, chain in ipairs(scan.tape_chains or {}) do
corpus.tape_chains[#corpus.tape_chains + 1] = chain
end
end
end
end
local SCHEMA_BODY_ERROR = {
reguse_malformed = true,
reguse_unknown_reg_type = true,
reguse_duplicate_alias = true,
reguse_duplicate_slot = true,
reguse_const_reg_spelling = true,
reguse_mixed_const = true,
}
-- Re-parse every RegUse_* body against the merged type_name_registry.
-- Scan-time expansion still runs when Reg_T is in the same source.
-- Missing Reg_T after merge is reguse_unknown_reg_type, not a fallback table.
local function resolve_reg_use_schemas(corpus)
local kept = {}
for _, err in ipairs(corpus.reg_use_errors or {}) do
if not SCHEMA_BODY_ERROR[err.kind] then
kept[#kept + 1] = err
end
end
corpus.reg_use_errors = kept
for name, type_entry in pairs(corpus.type_name_registry or {}) do
if name:sub(1, 7) == "RegUse_" and type_entry.body then
local fresh, errs = parse_reg_use_schema_body(
type_entry.body, corpus.type_name_registry, { require_types = true })
if fresh then
fresh.name = name
local old = corpus.reg_use_schemas[name]
fresh.source_file = (old and old.source_file) or type_entry.source_file
fresh.source_line = (old and old.source_line) or type_entry.source_line
corpus.reg_use_schemas[name] = fresh
else
corpus.reg_use_schemas[name] = nil
end
for _, err in ipairs(errs or {}) do
err.schema_name = name
err.source_file = type_entry.source_file
err.source_line = type_entry.source_line
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
end
end
end
end
@@ -2373,6 +2903,7 @@ function M.run(ctx)
-- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline).
merge_corpus_registries(corpus)
resolve_reg_use_schemas(corpus)
-- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return.
return { outputs = {}, errors = {}, warnings = {} }
+620 -86
View File
@@ -279,6 +279,16 @@ local function classify_tokens(tokens)
for tok_idx, t in ipairs(tokens) do
local tok = t.tok
local ident = tok:match("^([%w_]+)") or "?"
local is_delay_marker = false
local delay_marker = nil
if duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[ident] then
is_delay_marker = true
delay_marker = ident
local rest = tok:match("^[%w_]+%s+(.*)$")
if rest and rest ~= "" then
ident = rest:match("^([%w_]+)") or ident
end
end
local nop_words = 0
if ident == "nop" then nop_words = 1
elseif ident == "nop2" then nop_words = 2 end
@@ -330,7 +340,9 @@ local function classify_tokens(tokens)
local shape = ident:match("^mac_format_([%w_]+)_color$")
if shape then mac_format_shape = shape end
if ident:match("^mac_gte_store_[%w_]+$") then is_gte_store = true end
if ident:match("^mac_insert_ot_tag_[%w_]+$") then is_ot_tag = true end
if ident == "mac_insert_ot_tag" or ident:match("^mac_insert_ot_tag_[%w_]+$") then
is_ot_tag = true
end
-- O_(<arg1>, <arg2>) / S_(<arg>) captures (used by check_abi_handoff).
-- Cheap pattern match — anchored, fails fast on non-matching tokens.
@@ -343,6 +355,8 @@ local function classify_tokens(tokens)
tc[tok_idx] = {
ident = ident,
is_delay_marker = is_delay_marker,
delay_marker = delay_marker,
nop_words = nop_words,
nop_prefix = nop_run,
is_yield = is_yield,
@@ -451,6 +465,14 @@ local function is_cop2_consumer_of(consumer_event, destination, producer_rel)
return false
end
local function gpr_identity(event, pos)
local keys = event and event.gpr_keys
if keys and keys[pos] then return keys[pos] end
local arg = event and event.args and event.args[pos]
if type(arg) == "string" and arg:sub(1, 2) == "R_" then return arg end
return nil
end
-- True iff `consumer_event` reads the GPR operand at any position the destination register occupies.
-- read_pos lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
local function is_gpr_consumer_of(consumer_event, destination)
@@ -458,9 +480,8 @@ local function is_gpr_consumer_of(consumer_event, destination)
local read_pos = duffle.OPERAND_READ_POSITIONS or {}
local positions = read_pos[consumer_token]
if not positions then return false end
local args = consumer_event.args or {}
for _, pos in ipairs(positions) do
if args[pos] == destination then return true end
if gpr_identity(consumer_event, pos) == destination then return true end
end
return false
end
@@ -552,6 +573,11 @@ local function is_gpr_operand(operand)
return type(operand) == "string" and operand:sub(1, 2) == "R_"
end
local function is_tracked_gpr(operand)
return is_gpr_operand(operand)
or (type(operand) == "string" and operand:sub(1, 7) == "reguse:")
end
local function constant_for_operand(gpr_values, operand)
if operand == "R_0" then return 0 end
local slot = is_gpr_operand(operand) and gpr_values[operand] or nil
@@ -560,13 +586,13 @@ local function constant_for_operand(gpr_values, operand)
end
local function invalidate_gpr(gpr_values, operand)
if is_gpr_operand(operand) and operand ~= "R_0" then
if is_tracked_gpr(operand) and operand ~= "R_0" then
gpr_values[operand] = { kind = "unknown" }
end
end
local function store_gpr_constant(gpr_values, operand, value)
if not is_gpr_operand(operand) or operand == "R_0" then return end
if not is_tracked_gpr(operand) or operand == "R_0" then return end
if value == nil then gpr_values[operand] = { kind = "unknown" }
else gpr_values[operand] = { kind = "constant", value = wrap_u4(value) }
end
@@ -629,26 +655,31 @@ end
-- Encoders without an explicit effect row conservatively invalidate every R_-prefixed operand.
-- Recognized value rules are evaluated before their destination is invalidated.
-- A failed/unknown evaluation writes `{kind = "unknown"}` instead.
local function apply_gpr_effects(ev_ident, ev_args, forward_state)
local function apply_gpr_effects(ev, forward_state)
local ev_ident = ev.encoder or ev.ident
local ev_args = ev.args or {}
local gpr_values = forward_state.gpr_values
local effects = duffle.INSTRUCTION_GPR_EFFECTS or {}
local row = effects[ev_ident]
if row == nil then
for _, operand in ipairs(ev_args or {}) do
invalidate_gpr(gpr_values, operand)
for pos, operand in ipairs(ev_args) do
local key = gpr_identity(ev, pos) or operand
if type(key) == "string" and (key:sub(1, 2) == "R_" or key:sub(1, 7) == "reguse:") then
if key ~= "R_0" then gpr_values[key] = { kind = "unknown" } end
end
end
return
end
local value_rule = (duffle.GPR_VALUE_RULES or {})[ev_ident]
local value = value_rule and evaluate_gpr_value_rule(value_rule, ev_args or {}, gpr_values) or nil
local value = value_rule and evaluate_gpr_value_rule(value_rule, ev_args, gpr_values) or nil
for _, position in ipairs(row.writes or {}) do
local destination = ev_args and ev_args[position]
if is_gpr_operand(destination) then
local destination = gpr_identity(ev, position)
if destination then
if value_rule and position == value_rule.dest and value ~= nil then
store_gpr_constant(gpr_values, destination, value)
else
invalidate_gpr(gpr_values, destination)
if destination ~= "R_0" then gpr_values[destination] = { kind = "unknown" } end
end
end
end
@@ -944,6 +975,7 @@ local function analyze_hardware_relations(atom)
semantic = relation.semantic,
producer_word = prod.word,
consumer_word = ev_word,
destination = prod.destination,
gap = gap,
required = prod.required,
satisfied = satisfied,
@@ -953,7 +985,7 @@ local function analyze_hardware_relations(atom)
end
-- ── 2. Apply GPR value effects. ──
apply_gpr_effects(ev_ident, ev_args, forward)
apply_gpr_effects(ev, forward)
-- ── 3. Stage producers created by this event. ──
local rows = rows_by_token[ev_ident]
@@ -962,7 +994,7 @@ local function analyze_hardware_relations(atom)
-- `stage = false` rows document a direction but do not create a later command-input producer (SWC2 and ordinary MTC0).
if row.stage ~= false then
local dest_arg = row.writes and row.writes.arg
local destination = dest_arg and ev_args[dest_arg] or nil
local destination = dest_arg and (gpr_identity(ev, dest_arg) or ev_args[dest_arg]) or nil
if destination then
-- Apply the destination_match filter when present.
if row.destination_match and row.destination_match ~= destination then
@@ -1305,8 +1337,14 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
if is_load_delay then
-- Determine the destination register from the load's `writes` field.
local prev_writes = gpr_effects[prev_ident] and gpr_effects[prev_ident].writes or {}
local prev_args = prev_ev.args or {}
local load_dest = prev_writes[1] and prev_args[prev_writes[1]] or "<load-destination>"
local dest_pos = prev_writes[1]
local load_dest = dest_pos and (gpr_identity(prev_ev, dest_pos) or (prev_ev.args or {})[dest_pos]) or "<load-destination>"
local authored = dest_pos and (prev_ev.args or {})[dest_pos] or load_dest
local shown = authored
if type(load_dest) == "string" and load_dest:sub(1, 7) == "reguse:" then
local slot = load_dest:match("([^:]+)$")
if slot then shown = authored .. " (slot " .. slot .. ")" end
end
findings[#findings + 1] = {
check = "hazard_nop_use",
kind = "info",
@@ -1319,7 +1357,7 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
producer_destination = load_dest,
consumer_token = "<would-be-consumer>",
msg = string.format("%s at line %d: nop at word %d is modeled-required (load-delay slot for %s)"
, atom.name, ev_line, ev_word, load_dest
, atom.name, ev_line, ev_word, shown
),
}
else
@@ -1352,7 +1390,7 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
for _, row in ipairs(relations_table) do
if row.token == ev_ident and row.stage ~= false then
local dest_arg = row.writes and row.writes.arg
local destination = dest_arg and ev_args[dest_arg] or nil
local destination = dest_arg and (gpr_identity(ev, dest_arg) or ev_args[dest_arg]) or nil
if destination and (not row.destination_match or row.destination_match == destination) then
local required = row.visibility and row.visibility.required
if required == nil and not (row.visibility and row.visibility.kind == "unknown_consumer") then
@@ -1517,11 +1555,12 @@ local function check_load_delay_slots(atom, pipe_ctx, findings)
-- Use `net_reads` to ignore RMW positions (write shadows read within the same instruction).
if not is_load then
for _, pos in ipairs(net_reads(event_ident, args)) do
local reg = args[pos]
if type(reg) == "string" and reg:sub(1, 2) == "R_" then
local reg = gpr_identity(event, pos)
if reg then
local until_idx = volatile_until[reg]
if until_idx and event_idx <= until_idx then
local ev_line = line_for_word_event(event)
local authored = args[pos] or reg
findings[#findings + 1] = {
atom = atom.name,
line = ev_line,
@@ -1530,7 +1569,7 @@ local function check_load_delay_slots(atom, pipe_ctx, findings)
msg = string.format("%s at line %d reads %s at word %d, but a prior load's "
.. "delay slot is not over until word %d; insert a `nop` between the "
.. "load and this instruction.",
atom.name, ev_line, reg, event_idx, until_idx),
atom.name, ev_line, authored, event_idx, until_idx),
}
end
end
@@ -1541,8 +1580,8 @@ local function check_load_delay_slots(atom, pipe_ctx, findings)
local effect = gpr_effects[event_ident]
if effect and effect.writes then
for _, pos in ipairs(effect.writes) do
local reg = args[pos]
if type(reg) == "string" and reg:sub(1, 2) == "R_" then
local reg = gpr_identity(event, pos)
if reg then
if is_load then
-- Load: destination volatile for exactly 1 slot (the delay slot).
volatile_until[reg] = event_idx + 1
@@ -1673,9 +1712,10 @@ end
--- and the tape runtime would jump to garbage.
---
--- Rules:
--- 1. Every `mac_yield_load()` must be in a branch BD-slot (the immediately preceding token must be a branch).
--- 2. Every `mac_yield_tail()` must be the first instruction after an `atom_label()`, AND
--- at least one branch targeting that label must have `mac_yield_load()` in its BD-slot.
--- 1. Every `mac_yield_load()` must be in a branch BD-slot, or sit between two `atom_label`s.
--- Delay-marker prefixes are skipped when reading prev/next tokens.
--- 2. `mac_yield_tail()` is valid if every path that reaches it has already executed a `mac_yield_load()`.
--- A load in a branch BD slot always runs. Later branches that target the tail label may carry `nop`.
--- 3. `mac_yield_tail()` as the atom-end terminator (last token) is a WARNING, not an error
--- (the safe default for atom-endings is `mac_yield()` which re-loads `R_AtomJmp`).
---
@@ -1694,53 +1734,118 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
return atom.line + line_in_body[tokens[idx].rel]
end
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s (natural fall-through load pattern).
-- When the pattern is satisfied, the check stays silent; only violations emit findings.
local function is_delay_only(c)
return c and duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[c.ident] == true
end
local function skip_delay(idx, step)
local i = idx
while i >= 1 and i <= n and is_delay_only(tc[i]) do
i = i + step
end
if i < 1 or i > n then return nil end
return i
end
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s.
for tok_idx = 1, n do
local c = tc[tok_idx]
if c.ident == "mac_yield_load" then
local prev_tc = (tok_idx >= 2) and tc[tok_idx - 1] or nil
-- Look for the next `atom_label()` token (skip `atom_offset` markers; check immediately-adjacent first).
local next_label_tc = (tok_idx + 1 <= n) and tc[tok_idx + 1] or nil
if next_label_tc and next_label_tc.ident ~= "atom_label" then
next_label_tc = nil
for j = tok_idx + 1, n do
local t = tc[j]
if t.ident == "atom_label" then
next_label_tc = t
break
end
local prev_i = skip_delay(tok_idx - 1, -1)
local prev_tc = prev_i and tc[prev_i] or nil
local next_label_tc = nil
local j = skip_delay(tok_idx + 1, 1)
while j do
local t = tc[j]
if t.ident == "atom_label" then
next_label_tc = t
break
end
if t.ident ~= "atom_offset" then break end
j = skip_delay(j + 1, 1)
end
local natural_fallthrough = prev_tc and prev_tc.is_atom_label and next_label_tc ~= nil
if not natural_fallthrough then
if tok_idx < 2 or not prev_tc.is_branch then
if not prev_tc or not prev_tc.is_branch then
local prev_ident = prev_tc and (prev_tc.ident or "?") or "<none>"
local next_ident = next_label_tc and (next_label_tc.ident .. "(" .. (next_label_tc.label_name or "?") .. ")") or "<no following label>"
findings[#findings + 1] = {
atom = atom.name,
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
line = prev_i and line_for(tok_idx) or atom.line,
check = "yield_load_tail_pairing",
kind = "error",
msg = string.format(
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — and the next `atom_label()` token is `%s` — `mac_yield_load()` must fill a branch BD-slot or sit between two `atom_label`s for the natural fall-through load."
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
, atom.name, prev_i and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
}
end
end
end
end
-- ── Rule 2: every `mac_yield_tail()` must be at a labeled target whose branch BD-slot is `mac_yield_load()`.
-- ── Rule 2: `mac_yield_tail()` is valid if every path that reaches it already ran `mac_yield_load()`.
-- MIPS delay slot always runs. Successors skip the BD token for control flow,
-- but the yield walk still counts that token as executed.
local function load_covers_tail(tail_idx)
local labels = {}
for i = 1, n do
if tc[i].is_atom_label and tc[i].label_name then
labels[tc[i].label_name] = i
end
end
local function is_load(idx)
return tc[idx] and tc[idx].ident == "mac_yield_load"
end
local reached_without = false
local reached_any = false
local path_n = 0
local MAX_PATHS = 64
local function dfs(idx, saw_load, visited)
if path_n >= MAX_PATHS then return end
if visited[idx] then return end
local vis = {}
for k, v in pairs(visited) do vis[k] = v end
vis[idx] = true
local saw = saw_load or is_load(idx)
if tc[idx].is_branch and idx + 1 <= n then
saw = saw or is_load(idx + 1)
end
if idx == tail_idx then
path_n = path_n + 1
reached_any = true
if not saw then reached_without = true end
return
end
if tc[idx].is_yield or tc[idx].is_terminal_jump then
return
end
if tc[idx].is_branch then
if not tc[idx].is_unconditional_jump and idx + 2 <= n then
dfs(idx + 2, saw, vis)
end
local label = tc[idx].branch_label
if label and labels[label] then
local dest = labels[label] + 1
if dest <= n then dfs(dest, saw, vis) end
end
return
end
if idx + 1 <= n then
dfs(idx + 1, saw, vis)
end
end
dfs(1, false, {})
if not reached_any then return true end
return not reached_without
end
for tok_idx = 1, n do
local c = tc[tok_idx]
if c.ident ~= "mac_yield_tail" then goto continue end
-- The immediately preceding token must be an `atom_label()` (no instructions between them).
local prev_idx = tok_idx - 1
if prev_idx < 1 or not tc[prev_idx].is_atom_label then
local prev_idx = skip_delay(tok_idx - 1, -1)
if not prev_idx or not tc[prev_idx].is_atom_label then
if tok_idx == n then
-- Atom-ending case: last token is `mac_yield_tail()` without a preceding label. WARNING.
findings[#findings + 1] = {
atom = atom.name,
line = line_for(tok_idx),
@@ -1765,36 +1870,14 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
end
local label_name = tc[prev_idx].label_name
-- Find at least one branch targeting `label_name` whose BD-slot is `mac_yield_load()`.
local found_pairing = false
for branch_idx = 1, n do
local bt = tc[branch_idx]
if bt.is_branch and bt.branch_label == label_name then
local bd_idx = branch_idx + 1
local bd_tc = bd_idx <= n and tc[bd_idx] or nil
if bd_tc and bd_tc.ident == "mac_yield_load" then
found_pairing = true
else
findings[#findings + 1] = {
atom = atom.name,
line = line_for(branch_idx),
check = "yield_load_tail_pairing",
kind = "error",
msg = string.format(
"%s at line %d has `mac_yield_tail()` at label `%s` (word %d) but the branch targeting it (at word %d) has BD-slot `%s` instead of `mac_yield_load()`."
, atom.name, line_for(branch_idx), label_name, tok_idx, branch_idx, bd_tc and bd_tc.ident or "?"),
}
end
end
end
if not found_pairing then
if not load_covers_tail(tok_idx) then
findings[#findings + 1] = {
atom = atom.name,
line = line_for(tok_idx),
check = "yield_load_tail_pairing",
kind = "error",
msg = string.format(
"%s at line %d has `mac_yield_tail()` at label `%s` but no branch in the body targets this label with `mac_yield_load()` in its BD-slot — R_AtomJmp would not be loaded."
"%s at line %d has `mac_yield_tail()` at label `%s` but no path that reaches it has executed `mac_yield_load()` — R_AtomJmp would not be loaded."
, atom.name, line_for(tok_idx), label_name),
}
end
@@ -1912,6 +1995,7 @@ local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
local contrib = 0
local saw_format = false
local saw_prim_write = false
local saw_tag = false
-- Reads from tc_entry fields pre-computed by classify_tokens (R3 lift).
-- Eliminates 4 per-token string matches (mac_format_X_color + mac_gte_store_<shape> + mac_insert_ot_tag_<shape> + R_PrimCursor)
@@ -1942,10 +2026,57 @@ local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
local comp = pipe_ctx.components_by_name[bare]
local n = comp and comp.gp0_contrib
if n then contrib = contrib + n end
-- insert_ot_tag writes the packet tag. Count it once when the atom
-- has no raw O_(Poly_*, tag) store.
if not saw_tag then
contrib = contrib + 1
saw_tag = true
end
end
if tc_entry.writes_r_prim_cursor then
saw_prim_write = true
end
-- A raw store to O_(Poly_*, tag) is the packet tag word, counted once.
if tc_entry.o_arg1 and tc_entry.o_arg1:match("^Poly_") and tc_entry.o_arg2 == "tag" then
if tc_entry.is_store_word or tc_entry.ident == "gte_sw" then
if not saw_tag then
contrib = contrib + 1
saw_tag = true
end
end
end
end
-- Token-name gp0_contrib is 0 when bodies use uncounted stores.
-- Once gte_sw is taught, prefer that sum. Do not also count every PrimCursor store.
if contrib == 0 then
local seen_field = {}
for _, ev in ipairs(atom.paths.word_events or {}) do
local enc = ev.encoder or ""
if enc == "store_word" or enc == "store_half" or enc == "store_byte" or enc == "gte_sw" then
local text = (ev.call_text or "") .. " " .. (ev.root_call_text or "")
if text:find("insert_ot_tag", 1, true) then
-- OT list mutation, not a packet word.
else
local field = text:match("O_%(([^%)]+)%)") or text
if not seen_field[field] then
local hit = text:find("R_PrimCursor", 1, true)
if not hit then
for _, arg in ipairs(ev.args or {}) do
if tostring(arg):find("R_PrimCursor", 1, true) then
hit = true
break
end
end
end
if hit then
seen_field[field] = true
contrib = contrib + 1
end
end
end
end
end
end
if not cmd_byte then
@@ -2024,7 +2155,9 @@ local function analyze_atom_paths(atom, pipe_ctx)
local c = tc[tok_idx]
local ident = c.ident
local cost
if ident:sub(1, #"mac_") == "mac_" then
if duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[ident] then
cost = 0
elseif ident:sub(1, #"mac_") == "mac_" then
-- `mac_*` token: lookup corpus.components[bare_name].cycle_cost.
local bare = ident:sub(#"mac_" + 1)
local comp = pipe_ctx.components_by_name and pipe_ctx.components_by_name[bare]
@@ -2203,13 +2336,11 @@ end
-- Per-source rule (called once per source via the CHECK_RULES dispatch).
-- Signature matches the per_source shape established by check_semantic_reg_defaults.
--
-- Severity: WARNING (build continues).
-- The rule is intentionally permissive because the production `code/duffle/` and `code/gte_hello/`
-- sources use R_* aliases in atom_reads / atom_writes that may not yet be opted in via the bare `atom_reg` marker.
-- R_TapePtr / R_AtomJmp / R_PrimCursor / R_FaceCursor / R_VertBase / R_OtBase ARE opted in.
-- Raw C-ABI aliases like R_T0..R_T3 require explicit opt-in; the prototype keeps wave-context registration explicit.
-- no auto-include of wave-context; explicit opt-in only).
-- Warnings keep the build green and report aliases that need explicit registration.
-- Physical GPRs are members. Opt-in atom_reg stays for aliases.
local function is_physical_gpr(reg)
return type(reg) == "string" and (reg:match("^R_T[0-7]$") ~= nil or reg:match("^R_V[01]$") ~= nil)
end
local function check_enum_alias_membership(_src, pipe_ctx, findings)
local reg_registry = pipe_ctx.register_alias_registry or {}
@@ -2247,7 +2378,7 @@ local function check_enum_alias_membership(_src, pipe_ctx, findings)
end
end
for _, reg in ipairs(ai.reads or {}) do
if not reg_registry[reg] then
if not reg_registry[reg] and not is_physical_gpr(reg) then
findings[#findings + 1] = {
atom = atom_name, line = info_line,
check = "enum_alias_membership", kind = "warning",
@@ -2257,7 +2388,7 @@ local function check_enum_alias_membership(_src, pipe_ctx, findings)
end
end
for _, reg in ipairs(ai.writes or {}) do
if not reg_registry[reg] then
if not reg_registry[reg] and not is_physical_gpr(reg) then
findings[#findings + 1] = {
atom = atom_name, line = info_line,
check = "enum_alias_membership", kind = "warning",
@@ -2328,6 +2459,18 @@ local function find_field_by_name(type_entry, field_name)
return nil
end
-- Walk typedef aliases to the struct that owns the fields table. Depth matches propagate_type_sizes.
local function resolve_type_with_fields(type_name, type_registry, depth)
if depth > 8 then return nil end
local entry = type_registry[type_name]
if not entry then return nil end
if entry.fields then return entry end
if entry.kind == "typedef" and entry.underlying_type and entry.underlying_type ~= "" then
return resolve_type_with_fields(entry.underlying_type, type_registry, depth + 1)
end
return entry
end
-- True iff a (field, type_registry) pair is a leaf scalar (safe to dereference as a tape-payload field).
-- Pointer-to-X is always a leaf; non-pointer struct members fail the leaf test.
local function is_field_leaf(field, type_registry)
@@ -2355,8 +2498,18 @@ local function check_binds_no_substruct_deref(_src, pipe_ctx, findings)
local field_name = tc_entry.o_arg2
local body_line = a.line + (line_in_body[tokens[ti].rel] or 0)
local type_entry = type_registry[type_name]
if not type_entry or not type_entry.fields then
local type_entry = resolve_type_with_fields(type_name, type_registry, 1)
local no_fields = not type_entry or not type_entry.fields or #type_entry.fields == 0
local raw_entry = type_registry[type_name]
local is_typedef_to_struct = raw_entry
and raw_entry.kind == "typedef"
and raw_entry.underlying_type
and type_registry[raw_entry.underlying_type]
and type_registry[raw_entry.underlying_type].fields
local skip_opaque = no_fields and not is_typedef_to_struct
if skip_opaque then
-- leave this token
elseif not type_entry or not type_entry.fields then
findings[#findings + 1] = {
atom = a.name, line = body_line,
check = "binds_no_substruct_deref", kind = "warning",
@@ -2426,6 +2579,44 @@ local function atom_body_token_source_line(atom, token, line_in_body)
return (atom.line or 0) + body_line - 1
end
local function ctrl_alias_from_text(text)
return tostring(text or ""):match("gte_cr_[%w_]+")
end
local function ctrl_writes_in_atom(atom)
local out = {}
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
if (ev.encoder or "") == "gte_mv_to_ctrl_r" then
local alias = ev.args and ev.args[2]
if type(alias) ~= "string" or not alias:match("^gte_cr_") then
alias = ctrl_alias_from_text(ev.call_text) or ctrl_alias_from_text(ev.root_call_text)
end
local src = ev.args and ev.args[1]
if type(src) == "string" then src = src:match("[%w_]+") end
if alias then
out[#out + 1] = {
alias = alias,
src = src,
line = ev.line or atom.line,
}
end
end
end
if #out == 0 then
for _, t in ipairs((atom.paths and atom.paths.tokens) or {}) do
local tok = t.tok or ""
if tok:match("^gte_mv_to_ctrl_r") then
local alias = ctrl_alias_from_text(tok)
local src = tok:match("%(%s*([%w_]+)")
if alias then
out[#out + 1] = { alias = alias, src = src, line = atom.line }
end
end
end
end
return out
end
-- Check #N: gte_cr_alias_writes
-- Fires one warning per atom per alias-group when the atom body touches two
-- distinct aliases from the same group. Aliases within a group write to the
@@ -2566,6 +2757,173 @@ local function check_gte_cr_TR_naming(atom, _pipe_ctx, findings)
end
end
local function check_gte_cr_alias_writes_xatom(src, pipe_ctx, findings)
-- Walk tape chains once (first source only). Atoms in no chain stay per-atom.
local first = pipe_ctx.source_order and pipe_ctx.source_order[1]
if first and src ~= first then return end
local atoms_by_name = pipe_ctx.atoms_by_name or {}
for _, chain in ipairs(pipe_ctx.tape_chains or {}) do
local slot_state = {}
for _, name in ipairs(chain) do
local atom = atoms_by_name[name]
if atom then
atom.paths = atom.paths or {}
atom.paths.forward_state = atom.paths.forward_state or {}
local outgoing = {}
for slot, prev in pairs(slot_state) do
outgoing[slot] = prev
end
for _, w in ipairs(ctrl_writes_in_atom(atom)) do
local group = find_alias_pair_for(w.alias, duffle)
if group then
local slot = group[1]
local prev = slot_state[slot]
if prev and prev.alias ~= w.alias and prev.atom ~= atom.name then
findings[#findings + 1] = {
atom = atom.name or "",
line = w.line,
check = "gte_cr_alias_writes_xatom",
kind = "warning",
msg = string.format(
"atom '%s' writes %s to C2[%d]; atom '%s' already wrote %s"
, atom.name or "", w.alias, slot, prev.atom, prev.alias),
}
end
slot_state[slot] = { alias = w.alias, atom = atom.name, line = w.line }
outgoing[slot] = slot_state[slot]
end
end
atom.paths.forward_state.ctrl_writes_by_slot = outgoing
end
end
end
end
local function check_gte_packed_writes(atom, _pipe_ctx, findings)
local writes = ctrl_writes_in_atom(atom)
local first_idx = {}
for i, w in ipairs(writes) do
if first_idx[w.alias] == nil then first_idx[w.alias] = i end
end
for _, rel in ipairs(duffle.GTE_PACKED_SLOT_RELATIONS or {}) do
local i1 = first_idx[rel.first]
local i2 = first_idx[rel.second]
if i1 and i2 and i2 < i1 then
findings[#findings + 1] = {
atom = atom.name or "",
line = writes[i2].line,
check = "gte_packed_writes",
kind = "warning",
msg = string.format(
"atom '%s' writes %s before %s on packed C2[%d]"
, atom.name or "", rel.second, rel.first, rel.slot),
}
end
end
end
local function check_ctc2_chain_source_preservation(atom, _pipe_ctx, findings)
-- Fire only when a load sits before a later RT ctc2 that still names that GPR.
-- A load after the last RT ctc2 and before the command is a legal reload.
local events = (atom.paths and atom.paths.word_events) or {}
local function event_src(ev)
if ev.gpr_keys and ev.gpr_keys[1] then return ev.gpr_keys[1] end
local src = ev.args and ev.args[1]
if type(src) == "string" then src = src:match("^[%w_.]+") end
return src
end
local function event_alias(ev)
local alias = ev.args and ev.args[2]
if type(alias) ~= "string" or not alias:match("^gte_cr_") then
alias = ctrl_alias_from_text(ev.call_text)
end
return alias
end
if #events > 0 then
for i, ev in ipairs(events) do
local enc = ev.encoder or ""
if enc == "load_word" then
local dest = event_src(ev)
if dest then
local earlier = false
for j = i - 1, 1, -1 do
local prev = events[j]
local prev_enc = prev.encoder or ""
if prev_enc:match("^gte_cmdw_") then break end
if prev_enc == "gte_mv_to_ctrl_r" then
local prev_src = event_src(prev)
local prev_alias = event_alias(prev)
if prev_src == dest and prev_alias and prev_alias:match("^gte_cr_RT") then
earlier = true
break
end
end
end
if earlier then
for j = i + 1, #events do
local later = events[j]
local later_enc = later.encoder or ""
if later_enc:match("^gte_cmdw_") then
break
end
if later_enc == "gte_mv_to_ctrl_r" then
local later_src = event_src(later)
local later_alias = event_alias(later)
if later_src == dest and later_alias and later_alias:match("^gte_cr_RT") then
findings[#findings + 1] = {
atom = atom.name or "",
line = ev.line or atom.line,
check = "ctc2_chain_source_preservation",
kind = "warning",
msg = string.format(
"atom '%s' reloads %s before a later ctc2 that still names it"
, atom.name or "", dest),
}
break
end
end
end
end
end
end
end
return
end
local tokens = (atom.paths and atom.paths.tokens) or {}
for i, t in ipairs(tokens) do
local tok = t.tok or ""
local ident = tok:match("^([%w_]+)") or ""
if ident == "load_word" then
local dest = tok:match("%(%s*([%w_]+)")
if dest then
for j = i + 1, #tokens do
local later = tokens[j].tok or ""
local later_ident = later:match("^([%w_]+)") or ""
if later_ident:match("^gte_cmdw_") then
break
end
if later_ident == "gte_mv_to_ctrl_r" then
local later_src = later:match("%(%s*([%w_]+)")
local later_alias = ctrl_alias_from_text(later)
if later_src == dest and later_alias and later_alias:match("^gte_cr_RT") then
findings[#findings + 1] = {
atom = atom.name or "",
line = atom.line,
check = "ctc2_chain_source_preservation",
kind = "warning",
msg = string.format(
"atom '%s' reloads %s before a later ctc2 that still names it"
, atom.name or "", dest),
}
break
end
end
end
end
end
end
end
-- check_immediate_field_width — flags integer literals passed to instruction
-- macros that exceed the immediate field width. Reads `IMMEDIATE_FIELD_WIDTHS`
-- from duffle.lua. Only fires on parseable integer literals; register names,
@@ -2646,15 +3004,176 @@ local function check_immediate_field_width(atom, pipe_ctx, findings)
"%s: immediate %d at arg %d overflows %d-bit unsigned field (valid 0..%d)",
ev_ident, value, rule.arg, width, field_max),
}
end
end
end
end
end
end
end
end
end
end
end
end
local SCRATCH_GPRS = {
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
R_AT = true, R_V0 = true, R_V1 = true,
}
local function token_arg_list(tok)
local inner = (tok or ""):match("%b()")
if not inner then return {} end
return duffle.split_top_level_commas(inner:sub(2, -2))
end
local function arg_as_gpr(arg)
arg = duffle.trim(arg or "")
return arg:match("^R_[%w_]+$")
end
local function collect_gpr_traffic(tokens)
local reads, writes = {}, {}
for _, t in ipairs(tokens or {}) do
local tok = t.tok or t
local ident = (tok or ""):match("^([%w_]+)") or ""
if ident:sub(1, 4) ~= "mac_"
and not (duffle.DELAY_MARKERS and duffle.DELAY_MARKERS[ident])
and ident ~= "nop" and ident ~= "atom_label" and ident ~= "atom_offset"
then
local args = token_arg_list(tok)
local fx = (duffle.INSTRUCTION_GPR_EFFECTS or {})[ident]
if fx then
for _, pos in ipairs(fx.reads or {}) do
local g = arg_as_gpr(args[pos])
if g then reads[g] = true end
end
for _, pos in ipairs(fx.writes or {}) do
local g = arg_as_gpr(args[pos])
if g then writes[g] = true end
end
else
for _, arg in ipairs(args) do
local g = arg_as_gpr(arg)
if g then
reads[g] = true
writes[g] = true
end
end
end
end
end
return reads, writes
end
local function gpr_set_from_list(list)
local s = {}
for _, name in ipairs(list or {}) do
if type(name) == "string" then s[name] = true end
end
return s
end
local function gpr_set_eq(a, b)
for k in pairs(a) do if not b[k] then return false end end
for k in pairs(b) do if not a[k] then return false end end
return true
end
local function gpr_set_keys(s)
local keys = {}
for k in pairs(s) do keys[#keys + 1] = k end
table.sort(keys)
return keys
end
local function check_atom_calls_inferred_traffic(atom, pipe_ctx, findings)
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
if is_runtime_helper(atom) then return end
local info = pipe_ctx.info_by_atom and pipe_ctx.info_by_atom[atom.name]
if not info then return end
if #(info.reads or {}) == 0 and #(info.writes or {}) == 0 then return end
local tokens = (atom.paths and atom.paths.tokens) or atom.body_tokens
local reads, writes = collect_gpr_traffic(tokens)
for _, t in ipairs(tokens or {}) do
local ident = ((t.tok or t) or ""):match("^([%w_]+)") or ""
if ident:sub(1, 4) == "mac_" then
local bare = ident:sub(5)
local idx = pipe_ctx.component_body_index and pipe_ctx.component_body_index[bare]
local comp = (pipe_ctx.components_by_name or {})[bare]
or (pipe_ctx.atoms_by_name or {})[bare]
local body_toks = (idx and idx.body_tokens)
or (comp and (comp.body_tokens or (comp.paths and comp.paths.tokens)))
local cr, cw = collect_gpr_traffic(body_toks)
for k in pairs(cr) do reads[k] = true end
for k in pairs(cw) do writes[k] = true end
end
end
local decl_r = gpr_set_from_list(info.reads)
local decl_w = gpr_set_from_list(info.writes)
local function keep_inferred(inferred, declared)
local out = {}
for k in pairs(inferred) do
if k == "R_0" then
if declared[k] then out[k] = true end
elseif SCRATCH_GPRS[k] then
if declared[k] then out[k] = true end
else
out[k] = true
end
end
return out
end
reads = keep_inferred(reads, decl_r)
writes = keep_inferred(writes, decl_w)
if not gpr_set_eq(decl_r, reads) or not gpr_set_eq(decl_w, writes) then
findings[#findings + 1] = {
atom = atom.name,
line = info.info_line or atom.line,
check = "atom_calls_inferred_traffic",
kind = "warning",
msg = string.format(
"atom '%s' declared [%s]/[%s] != inferred [%s]/[%s]",
atom.name,
table.concat(gpr_set_keys(decl_r), ","),
table.concat(gpr_set_keys(decl_w), ","),
table.concat(gpr_set_keys(reads), ","),
table.concat(gpr_set_keys(writes), ",")),
}
end
end
local function check_component_self_consistency(src, pipe_ctx, findings)
local first = pipe_ctx.source_order and pipe_ctx.source_order[1]
if first and src ~= first then return end
local infos = pipe_ctx.component_atom_infos or {}
local atoms_by_name = pipe_ctx.atoms_by_name or {}
for _, ai in ipairs(infos) do
local name = ai.atom_name or ai.name
local atom = name and atoms_by_name[name]
if atom and not atom.debug_skip then
local tokens = (atom.paths and atom.paths.tokens) or atom.body_tokens
local reads, writes = collect_gpr_traffic(tokens)
local decl_r = gpr_set_from_list(ai.reads)
local decl_w = gpr_set_from_list(ai.writes)
if not gpr_set_eq(decl_r, reads) or not gpr_set_eq(decl_w, writes) then
findings[#findings + 1] = {
atom = name,
line = ai.info_line or atom.line,
check = "component_self_consistency",
kind = "warning",
msg = string.format(
"component '%s' atom_reads/atom_writes [%s]/[%s] != body [%s]/[%s]",
name,
table.concat(gpr_set_keys(decl_r), ","),
table.concat(gpr_set_keys(decl_w), ","),
table.concat(gpr_set_keys(reads), ","),
table.concat(gpr_set_keys(writes), ",")),
}
end
end
end
end
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
-- ════════════════════════════════════════════════════════════════════════════
@@ -2683,12 +3202,17 @@ local CHECK_RULES = {
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
{ name = "gte_cr_alias_writes", per_atom = check_gte_cr_alias_writes },
{ name = "gte_cr_alias_writes_xatom", per_source = check_gte_cr_alias_writes_xatom },
{ name = "gte_packed_writes", per_atom = check_gte_packed_writes },
{ name = "ctc2_chain_source_preservation", per_atom = check_ctc2_chain_source_preservation },
{ name = "rtdiagonal_completeness", per_atom = check_rtdiagonal_completeness },
{ name = "gte_cr_TR_naming", per_atom = check_gte_cr_TR_naming },
{ name = "immediate_field_width", per_atom = check_immediate_field_width },
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
{ name = "component_self_consistency", per_source = check_component_self_consistency },
{ name = "atom_calls_inferred_traffic", per_atom = check_atom_calls_inferred_traffic },
}
-- ════════════════════════════════════════════════════════════════════════════
@@ -2725,6 +3249,11 @@ local function build_corpus_pipe_ctx(ctx)
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
components_by_name = corpus.components or {},
atoms_by_name = corpus.atoms_by_name or {},
tape_chains = corpus.tape_chains or {},
source_order = corpus.source_order or {},
component_atom_infos = corpus.component_atom_infos or {},
atom_infos = corpus.atom_infos or {},
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
atom_infos_list = corpus.atom_infos or {},
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
@@ -2783,6 +3312,11 @@ local function validate(ctx, src, corpus_pipe_ctx)
type_name_registry = corpus_pipe_ctx.type_name_registry,
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
components_by_name = corpus_pipe_ctx.components_by_name,
atoms_by_name = corpus_pipe_ctx.atoms_by_name,
tape_chains = corpus_pipe_ctx.tape_chains,
source_order = corpus_pipe_ctx.source_order,
component_atom_infos = corpus_pipe_ctx.component_atom_infos,
atom_infos_all = corpus_pipe_ctx.atom_infos,
}
-- Shared cross-source component-body index is owned by the corpus (`corpus.component_body_index`, populated by `passes/components.lua`).
-- Per-atom checks consume the corpus-owned index directly.