mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-09-08 17:29:05 +00:00
Compare commits
16
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
86fe189b4e | ||
|
|
da007d342e | ||
|
|
5a4bfb1224 | ||
|
|
d4795cf9de | ||
|
|
e79c364b40 | ||
|
|
18b1d5a04b | ||
|
|
581b00b960 | ||
|
|
3faccfc283 | ||
|
|
1a0d417649 | ||
|
|
3301826f5c | ||
|
|
d9b9241e2c | ||
|
|
a16c727db2 | ||
|
|
8a825a59c7 | ||
|
|
f8b28be02e | ||
|
|
ffc66052f8 | ||
|
|
7764612325 |
Vendored
+43
@@ -0,0 +1,43 @@
|
|||||||
|
# Package and install the local VS Code Insiders extensions under .vscode/.
|
||||||
|
# Usage:
|
||||||
|
# .\install_extensions.ps1
|
||||||
|
# .\install_extensions.ps1 -SkipPackage
|
||||||
|
|
||||||
|
param([switch] $SkipPackage)
|
||||||
|
|
||||||
|
$path_vscode = $PSScriptRoot
|
||||||
|
$code_insiders = "C:\apps\Microsoft VS Code Insiders\bin\code-insiders.cmd"
|
||||||
|
if (-not (test-path -literalpath $code_insiders)) {
|
||||||
|
$found = get-command code-insiders -erroraction silentlycontinue
|
||||||
|
if ($found) { $code_insiders = $found.source }
|
||||||
|
}
|
||||||
|
|
||||||
|
if (-not (test-path -literalpath $code_insiders)) { throw "code-insiders not found. Install VS Code Insiders or add it to PATH." }
|
||||||
|
|
||||||
|
$extensions = @(
|
||||||
|
(join-path $path_vscode "tape-atom-syntax"),
|
||||||
|
(join-path $path_vscode "cozy-and-windy")
|
||||||
|
)
|
||||||
|
|
||||||
|
foreach ($extension in $extensions) {
|
||||||
|
$package_json = join-path $extension "package.json"
|
||||||
|
if (-not (test-path -literalpath $package_json)) { throw "missing $package_json" }
|
||||||
|
|
||||||
|
$manifest = get-content -literalpath $package_json -raw | convertfrom-json
|
||||||
|
$vsix = join-path $extension ("{0}-{1}.vsix" -f $manifest.name, $manifest.version)
|
||||||
|
|
||||||
|
if (-not $SkipPackage) {
|
||||||
|
if (-not $manifest.scripts.package) { throw "$package_json has no scripts.package" }
|
||||||
|
write-host "packaging $($manifest.displayName) ($($manifest.name)@$($manifest.version))"
|
||||||
|
& npm --prefix $extension run package
|
||||||
|
if ($LASTEXITCODE -ne 0) { throw "npm run package failed for $extension" }
|
||||||
|
}
|
||||||
|
|
||||||
|
if (-not (test-path -literalpath $vsix)) { throw "missing $vsix" }
|
||||||
|
|
||||||
|
write-host "installing $vsix"
|
||||||
|
& $code_insiders --install-extension $vsix --force
|
||||||
|
if ($LASTEXITCODE -ne 0) { throw "install failed for $vsix" }
|
||||||
|
}
|
||||||
|
|
||||||
|
write-host "done. reload the Insiders window (Developer: Reload Window)."
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -49,7 +49,7 @@ const DSL_KEYWORDS = new Set([
|
|||||||
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
"u1_r", "u2_r", "u4_r", "u8_r", "u1_v", "u2_v", "u4_v", "u8_v",
|
||||||
]);
|
]);
|
||||||
|
|
||||||
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_"]);
|
const DELAY_SLOT_KEYWORDS = new Set(["LdSlot_", "BdSlot_", "DmaSlot_", "GteDelay_"]);
|
||||||
|
|
||||||
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
const CONTROL_FLOW_PREFIXES = /^(?:branch_|jump_|call_)/;
|
||||||
|
|
||||||
|
|||||||
@@ -56,7 +56,7 @@
|
|||||||
"name": "support.function.duffle.annotation"
|
"name": "support.function.duffle.annotation"
|
||||||
},
|
},
|
||||||
"delay-slots": {
|
"delay-slots": {
|
||||||
"match": "\\b(LdSlot_|BdSlot_)\\b",
|
"match": "\\b(LdSlot_|BdSlot_|DmaSlot_|GteDelay_)\\b",
|
||||||
"name": "keyword.operator.duffle.delayslot"
|
"name": "keyword.operator.duffle.delayslot"
|
||||||
},
|
},
|
||||||
"types": {
|
"types": {
|
||||||
|
|||||||
@@ -105,6 +105,16 @@ test("document-local declarations override an empty workspace index", () => {
|
|||||||
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
|
assert.equal(byText(result, "mac_new_component")[0].type, "tapeComponentInstruction");
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test("delay slot markers share the tapeDelaySlot token", () => {
|
||||||
|
const source = "LdSlot_ nop, BdSlot_ nop, DmaSlot_ nop2, GteDelay_ nop";
|
||||||
|
const result = classifyDocument(source, "C:/x/code/duffle/gte.atom.c", createIndex());
|
||||||
|
|
||||||
|
assert.equal(byText(result, "LdSlot_")[0].type, "tapeDelaySlot");
|
||||||
|
assert.equal(byText(result, "BdSlot_")[0].type, "tapeDelaySlot");
|
||||||
|
assert.equal(byText(result, "DmaSlot_")[0].type, "tapeDelaySlot");
|
||||||
|
assert.equal(byText(result, "GteDelay_")[0].type, "tapeDelaySlot");
|
||||||
|
});
|
||||||
|
|
||||||
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
|
test("classifier returns ordered non-overlapping spans and partial malformed output", () => {
|
||||||
const source = "atom_reads(R_A /* broken";
|
const source = "atom_reads(R_A /* broken";
|
||||||
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
const result = classifyDocument(source, "C:/x/code/test.atom.c", createIndex());
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# pragma once
|
# pragma once
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
bios_init_pad_2 = 0x12,
|
bios_init_pad_2 = 0x12,
|
||||||
bios_start_pad_2 = 0x13,
|
bios_start_pad_2 = 0x13,
|
||||||
|
|||||||
@@ -148,6 +148,7 @@ typedef void Proc_(VoidFn) (void);
|
|||||||
#define null C_(U4, 0)
|
#define null C_(U4, 0)
|
||||||
#define nullptr C_(void*, 0)
|
#define nullptr C_(void*, 0)
|
||||||
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
||||||
|
#define OA_(type, aexpr) C_(U4, & C_(type*,0) aexpr)
|
||||||
#define OT_(field) O_(typeof_ptr(& field), field))
|
#define OT_(field) O_(typeof_ptr(& field), field))
|
||||||
#define S_(data) C_(U4, sizeof(data))
|
#define S_(data) C_(U4, sizeof(data))
|
||||||
|
|
||||||
|
|||||||
+141
-42
@@ -17,7 +17,7 @@
|
|||||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
@@ -35,15 +35,11 @@
|
|||||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||||
* ---------------------------------------------------------------------------*/
|
* ---------------------------------------------------------------------------*/
|
||||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
|
|
||||||
// - mac_yield_load() + mac_yield_tail():
|
|
||||||
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
|
|
||||||
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
|
|
||||||
#define mac_yield(...) \
|
#define mac_yield(...) \
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||||
, jump_reg( R_AtomJmp) \
|
, jump_reg( R_AtomJmp) \
|
||||||
, nop
|
, BdSlot_ nop
|
||||||
WORD_COUNT(mac_yield, 4)
|
WORD_COUNT(mac_yield, 4)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
@@ -55,9 +51,20 @@ WORD_COUNT(mac_yield_load, 1)
|
|||||||
#define mac_yield_tail(...) \
|
#define mac_yield_tail(...) \
|
||||||
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
||||||
, jump_reg( R_AtomJmp) \
|
, jump_reg( R_AtomJmp) \
|
||||||
, nop
|
, BdSlot_ nop
|
||||||
WORD_COUNT(mac_yield_tail, 3)
|
WORD_COUNT(mac_yield_tail, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_half_v3(tx, ty, tz, base, offset) \
|
||||||
|
load_half(tx, base, offset + OA_(U2,[0])) \
|
||||||
|
, load_half(ty, base, offset + OA_(U2,[1])) \
|
||||||
|
, load_half(tz, base, offset + OA_(U2,[2]))
|
||||||
|
WORD_COUNT(mac_load_half_v3, 3)
|
||||||
|
|
||||||
|
#define mac_load_v3s2(transfer, base, offset) \
|
||||||
|
mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_load_v3s2, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||||
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
|
load_half(rs_x, r_base, offset + O_(V3_S2,x)) \
|
||||||
@@ -71,26 +78,75 @@ WORD_COUNT(mac_load_v2s2, 2)
|
|||||||
WORD_COUNT(mac_store_v2s2, 2)
|
WORD_COUNT(mac_store_v2s2, 2)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
|
#define mac_load_word_v3(tx, ty, tz, base, offset) \
|
||||||
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \
|
load_word(tx, base, offset + OA_(U4,[0])) \
|
||||||
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \
|
, load_word(ty, base, offset + OA_(U4,[1])) \
|
||||||
, load_word( rs_z, r_base, offset + O_(V3_S4,z))
|
, load_word(tz, base, offset + OA_(U4,[2]))
|
||||||
|
WORD_COUNT(mac_load_word_v3, 3)
|
||||||
|
|
||||||
|
#define mac_load_v3s4(transfer, base, offset) \
|
||||||
|
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
WORD_COUNT(mac_load_v3s4, 3)
|
WORD_COUNT(mac_load_v3s4, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
#define mac_load_p3s4(transfer, base, offset) \
|
||||||
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
store_word(rt_x, base, offset + O_(V3_S4,x)) \
|
WORD_COUNT(mac_load_p3s4, 3)
|
||||||
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
|
|
||||||
, store_word(rt_z, base, offset + O_(V3_S4,z))
|
|
||||||
WORD_COUNT(mac_store_v3s4, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
|
#define mac_store_half_v3(tx, ty, tz, base, offset) \
|
||||||
sub_s(rds_x, rds_x, rt_x) \
|
store_half(tx, base, offset + OA_(U2,[0])) \
|
||||||
, sub_s(rds_y, rds_y, rt_y) \
|
, store_half(ty, base, offset + OA_(U2,[1])) \
|
||||||
, sub_s(rds_z, rds_z, rt_z)
|
, store_half(tz, base, offset + OA_(U2,[2]))
|
||||||
|
WORD_COUNT(mac_store_half_v3, 3)
|
||||||
|
|
||||||
|
#define mac_store_v3s2(transfer, base, offset) \
|
||||||
|
mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_store_v3s2, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_word_v3(tx, ty, tz, base, offset) \
|
||||||
|
store_word(tx, base, offset + OA_(U4,[0])) \
|
||||||
|
, store_word(ty, base, offset + OA_(U4,[1])) \
|
||||||
|
, store_word(tz, base, offset + OA_(U4,[2]))
|
||||||
|
WORD_COUNT(mac_store_word_v3, 3)
|
||||||
|
|
||||||
|
#define mac_store_v3s4(transfer, base, offset) \
|
||||||
|
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_store_v3s4, 3)
|
||||||
|
|
||||||
|
#define mac_store_p3s4(transfer, base, offset) \
|
||||||
|
mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset)
|
||||||
|
WORD_COUNT(mac_store_p3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_add_si_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
||||||
|
add_si(rt_x, base, O_(V3_S4,x)) \
|
||||||
|
, add_si(rt_y, base, O_(V3_S4,y)) \
|
||||||
|
, add_si(rt_z, base, O_(V3_S4,z))
|
||||||
|
WORD_COUNT(mac_add_si_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_sub_s_v3(dx, dy, dz, sx, sy, sz, tx, ty, tz) \
|
||||||
|
sub_s(dx, sx, tx) \
|
||||||
|
, sub_s(dy, sy, ty) \
|
||||||
|
, sub_s(dz, sz, tz)
|
||||||
|
WORD_COUNT(mac_sub_s_v3, 3)
|
||||||
|
|
||||||
|
#define mac_sub_v3s4(d, s, t) \
|
||||||
|
mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z)
|
||||||
WORD_COUNT(mac_sub_v3s4, 3)
|
WORD_COUNT(mac_sub_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_sub_s_v3_self(ds_x, ds_y, ds_z, tx, ty, tz) \
|
||||||
|
sub_s(ds_x, ds_x, tx) \
|
||||||
|
, sub_s(ds_y, ds_y, ty) \
|
||||||
|
, sub_s(ds_z, ds_z, tz)
|
||||||
|
WORD_COUNT(mac_sub_s_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_sub_v3s4_self(ds, t) \
|
||||||
|
mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z)
|
||||||
|
WORD_COUNT(mac_sub_v3s4_self, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||||
@@ -105,6 +161,33 @@ WORD_COUNT(mac_store_rects2, 4)
|
|||||||
, or_i_self( dst, u4_lo(imm))
|
, or_i_self( dst, u4_lo(imm))
|
||||||
WORD_COUNT(mac_load_word_imm, 2)
|
WORD_COUNT(mac_load_word_imm, 2)
|
||||||
|
|
||||||
|
#define mac_shift_aright_v3_self(dt_x, dt_y, dt_z, shift_amount) \
|
||||||
|
shift_aright(dt_x, dt_x, shift_amount) \
|
||||||
|
, shift_aright(dt_y, dt_y, shift_amount) \
|
||||||
|
, shift_aright(dt_z, dt_z, shift_amount)
|
||||||
|
WORD_COUNT(mac_shift_aright_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_v3s4_self(dt, shift) \
|
||||||
|
mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_v3s4_self, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
|
||||||
|
shift_aright_var(rd_v0, rs_v0, r_shift) \
|
||||||
|
, shift_aright_var(rd_v1, rs_v1, r_shift) \
|
||||||
|
, shift_aright_var(rd_v2, rs_v2, r_shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
|
||||||
|
shift_aright_var(rds_v0, rds_v0, r_shift) \
|
||||||
|
, shift_aright_var(rds_v1, rds_v1, r_shift) \
|
||||||
|
, shift_aright_var(rds_v2, rds_v2, r_shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_var_v3s4_self(ds, shift) \
|
||||||
|
mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3s4_self, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
||||||
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
||||||
@@ -112,6 +195,30 @@ WORD_COUNT(mac_load_word_imm, 2)
|
|||||||
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
||||||
WORD_COUNT(mac_load_tri_indices, 3)
|
WORD_COUNT(mac_load_tri_indices, 3)
|
||||||
|
|
||||||
|
#define mac_gte_mv_to_cr_diag_v3s4(v) \
|
||||||
|
gte_mv_to_ctrl_r(v.y, gte_cr_RT13) \
|
||||||
|
, gte_mv_to_ctrl_r(v.z, gte_cr_RT22) \
|
||||||
|
, gte_mv_to_ctrl_r(v.x, gte_cr_RT11)
|
||||||
|
WORD_COUNT(mac_gte_mv_to_cr_diag_v3s4, 3)
|
||||||
|
|
||||||
|
#define mac_gte_ld_ir123_v3s4(v) \
|
||||||
|
gte_mv_to_data_r(v.x, C2_IR1) \
|
||||||
|
, gte_mv_to_data_r(v.y, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(v.z, C2_IR3)
|
||||||
|
WORD_COUNT(mac_gte_ld_ir123_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_op_cross_v3s4(a, b) \
|
||||||
|
mac_gte_mv_to_cr_diag_v3s4(a) \
|
||||||
|
GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */ \
|
||||||
|
, mac_gte_ld_ir123_v3s4(b) \
|
||||||
|
GteDelay_ /* IR: second operand (b.xyz) */ \
|
||||||
|
, gte_cmdw_cross /* OP: MAC1/2/3 = a × b (S12.20) */ \
|
||||||
|
, mac_gte_mv_from_mac123_v3s4(a) \
|
||||||
|
GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */ \
|
||||||
|
, mac_shift_aright_v3s4_self(a, 12) /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
||||||
|
WORD_COUNT(mac_gte_op_cross_v3s4, 16)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_gte_store_f3(r_primitive_cursor) \
|
#define mac_gte_store_f3(r_primitive_cursor) \
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
||||||
@@ -125,19 +232,19 @@ WORD_COUNT(mac_gte_store_f3, 3)
|
|||||||
, add_u_self(R_AT, r_vert_base) \
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
, gte_mv_to_data_r(R_V0, C2_VXY0) \
|
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||||
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
|
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
|
||||||
, add_u_self(R_AT, r_vert_base) \
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||||
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
|
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
|
||||||
, add_u_self(R_AT, r_vert_base) \
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
, gte_mv_to_data_r(R_V0, C2_VXY2) \
|
, LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||||
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
||||||
|
|
||||||
@@ -162,11 +269,11 @@ WORD_COUNT(mac_gte_store_g4_p3, 1)
|
|||||||
WORD_COUNT(mac_gte_sqr_v3, 8)
|
WORD_COUNT(mac_gte_sqr_v3, 8)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop_slot) \
|
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, delay_slot) \
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1) \
|
gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||||
, nop_slot \
|
, delay_slot \
|
||||||
, gte_cmdw_sqr
|
, gte_cmdw_sqr
|
||||||
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
||||||
|
|
||||||
@@ -176,7 +283,7 @@ WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
|||||||
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||||
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||||
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||||
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
, GteDelay_ nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
||||||
, gte_cmdw_gpf \
|
, gte_cmdw_gpf \
|
||||||
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
||||||
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
||||||
@@ -184,7 +291,7 @@ WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
|||||||
, shift_aright_var(r_dx, r_dx, r_shift) \
|
, shift_aright_var(r_dx, r_dx, r_shift) \
|
||||||
, shift_aright_var(r_dy, r_dy, r_shift) \
|
, shift_aright_var(r_dy, r_dy, r_shift) \
|
||||||
, shift_aright_var(r_dz, r_dz, r_shift)
|
, shift_aright_var(r_dz, r_dz, r_shift)
|
||||||
WORD_COUNT(mac_gte_gpf_scale, 13)
|
WORD_COUNT(mac_gte_gpf_scale, 12)
|
||||||
|
|
||||||
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
|
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
|
||||||
load_word( r_t0, r_off, O_(V3_S4,x)) \
|
load_word( r_t0, r_off, O_(V3_S4,x)) \
|
||||||
@@ -204,25 +311,13 @@ WORD_COUNT(mac_trans_mt3s3s4, 6)
|
|||||||
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
||||||
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
||||||
|
|
||||||
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
|
|
||||||
shift_aright_var(rd_v0, rs_v0, r_shift) \
|
|
||||||
, shift_aright_var(rd_v1, rs_v1, r_shift) \
|
|
||||||
, shift_aright_var(rd_v2, rs_v2, r_shift)
|
|
||||||
WORD_COUNT(mac_shift_aright_var_v3, 3)
|
|
||||||
|
|
||||||
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
|
|
||||||
shift_aright_var(rds_v0, rds_v0, r_shift) \
|
|
||||||
, shift_aright_var(rds_v1, rds_v1, r_shift) \
|
|
||||||
, shift_aright_var(rds_v2, rds_v2, r_shift)
|
|
||||||
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
|
|
||||||
|
|
||||||
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
||||||
gte_mv_to_data_r(to_ir0, C2_IR0) \
|
gte_mv_to_data_r(to_ir0, C2_IR0) \
|
||||||
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
||||||
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
||||||
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
||||||
, LdSlot_ nop_slot1 \
|
, GteDelay_ nop_slot1 \
|
||||||
, LdSlot_ nop_slot2 \
|
, GteDelay_ nop_slot2 \
|
||||||
, gte_cmdw_gpf \
|
, gte_cmdw_gpf \
|
||||||
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||||
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||||
@@ -235,6 +330,10 @@ WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
|
|||||||
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||||
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
|
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
|
||||||
|
|
||||||
|
#define mac_gte_mv_from_mac123_v3s4(v) \
|
||||||
|
mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z)
|
||||||
|
WORD_COUNT(mac_gte_mv_from_mac123_v3s4, 3)
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
/* atom_dbg_skip */
|
||||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||||
mac_load_word_imm(reg_transfer, cmd) \
|
mac_load_word_imm(reg_transfer, cmd) \
|
||||||
|
|||||||
@@ -14,7 +14,7 @@
|
|||||||
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.h
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
@@ -25,7 +25,15 @@
|
|||||||
#pragma region duffle
|
#pragma region duffle
|
||||||
|
|
||||||
|
|
||||||
// --- atom: normalize_v3s4 (47 words) ---
|
// --- atom: example_atom_proc (10 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_example_atom_proc_skip 2
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_example_atom_proc_skip = _atom_offset_example_atom_proc_skip,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: build_normalize_v3s4 (67 words) ---
|
||||||
|
|
||||||
#define _atom_offset_aligned_done_srav_path 3
|
#define _atom_offset_aligned_done_srav_path 3
|
||||||
#define _atom_offset_srav_path_aligned_done 4
|
#define _atom_offset_srav_path_aligned_done 4
|
||||||
|
|||||||
@@ -44,7 +44,8 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
||||||
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ab, {
|
// TODO(Ed): Expose R_T1 as a r_t0, r_V0 as r_t2
|
||||||
|
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, Reg r_ot_base, Reg r_prim_cursor, U2 poly_size) MipsAtomComp_Proc_(ab, {
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||||
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||||
|
|||||||
+4
-6
@@ -68,6 +68,7 @@ enum {
|
|||||||
|
|
||||||
#define gp0_send(word) (HW_GP0[0] = (word))
|
#define gp0_send(word) (HW_GP0[0] = (word))
|
||||||
#define gp1_send(word) (HW_GP1[0] = (word))
|
#define gp1_send(word) (HW_GP1[0] = (word))
|
||||||
|
#define DmaSlot_ // Annotate an instruction as filling a CPU <-> Command DMA delay slot/s
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
* GP0 command byte constants + Layer 1 (GPU bitfield shifts)
|
||||||
@@ -418,14 +419,11 @@ typedef Struct_(PolyTag) {
|
|||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
/* DSL cast convention: every cast uses `C_()`, every pointer qualifier is `R_` (restrict) or `V_` (volatile).
|
|
||||||
* No raw C-style casts. RHS values are assumed to be `U4` — caller passes a `U4` directly. */
|
|
||||||
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
#define set_len(tag,v) (C_(PolyTag_R,tag)->len = u4_(v))
|
||||||
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
#define set_addr(tag,v) (C_(PolyTag_R,tag)->addr = u4_(v))
|
||||||
/* `set_code` is no longer in the new PolyTag design — the code byte lives in the primitive body
|
/* `set_code` is no longer in the new PolyTag design
|
||||||
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
* (e.g. `((Poly_F3*)(p))->code`), not in the tag.
|
||||||
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters,
|
* Use the typed primitive structs (Poly_F3, Poly_G4, etc.) and the `set_poly_*` setters, which set both the tag's length and the code. */
|
||||||
* which set both the tag's length and the code. */
|
|
||||||
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
#define get_len(tag) C_(U4,C_(PolyTag_R,tag)->len)
|
||||||
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
#define get_addr(tag) C_(U4,C_(PolyTag_R,tag)->addr)
|
||||||
|
|
||||||
@@ -572,7 +570,7 @@ enum {
|
|||||||
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
/* Default TPage value libpsyx's SetDefDrawEnv writes (matches the `li v1, 10; sh v1, 20(v0)` sequence at C11_only.elf:0x8001273C). */
|
||||||
gp0_tpage_default = 10,
|
gp0_tpage_default = 10,
|
||||||
|
|
||||||
/* TPage semi-transparency mode payload values (NOT bit positions). */
|
/* TPage semi-transparency mode payload values. */
|
||||||
gp0_tpage_semi_trans_none = 0x0,
|
gp0_tpage_semi_trans_none = 0x0,
|
||||||
gp0_tpage_semi_trans_alpha = 0x1,
|
gp0_tpage_semi_trans_alpha = 0x1,
|
||||||
gp0_tpage_semi_trans_add = 0x2,
|
gp0_tpage_semi_trans_add = 0x2,
|
||||||
|
|||||||
+133
-103
@@ -18,6 +18,42 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|||||||
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||||
})
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_mv_to_cr_diag_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_ctrl_r(v.y, gte_cr_RT13),
|
||||||
|
gte_mv_to_ctrl_r(v.z, gte_cr_RT22),
|
||||||
|
gte_mv_to_ctrl_r(v.x, gte_cr_RT11),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_ld_ir123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(v.x, C2_IR1),
|
||||||
|
gte_mv_to_data_r(v.y, C2_IR2),
|
||||||
|
gte_mv_to_data_r(v.z, C2_IR3),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── GTE OP cross product (a × b → a) ───
|
||||||
|
* Sets up RT diagonal from a.xyz, IR1/2/3 from b.xyz, fires OP,
|
||||||
|
* reads MAC1/2/3, shifts right 12 (S12.20 → S12.0 OuterProduct12), writes back to a.xyz.
|
||||||
|
* Composes the three sub-primitives (RT-load, IR-load, OP, MAC-read, shift)
|
||||||
|
* into one component for use by atoms that need the cross product inline.
|
||||||
|
*
|
||||||
|
* Output gpr (a) aliases source-A gpr; MAC read clobbers source-A's load targets,
|
||||||
|
* but by that point the RT load is complete and source A is dead.
|
||||||
|
* Pipeline: clobbers IR1..3, MAC1..3, RT11..33.
|
||||||
|
*
|
||||||
|
* The CPU→COP2 transfer chains (3 ctc2, 3 mtc2) require a 2-slot retirement gap,
|
||||||
|
* and the MFC2→GPR chain (3 mfc2) requires a 1-slot retirement gap, before the GPR can be read.
|
||||||
|
* The hazard nops are inlined below — same convention as ac_gte_gpf_scale — so any atom body inlining this component inherits them.
|
||||||
|
*
|
||||||
|
* Words: 18 (3 ctc2 + 2 nop + 3 mtc2 + 2 nop + 1 op + 3 mfc2 + 1 nop + 3 sra).
|
||||||
|
*/
|
||||||
|
FI_ Slice_MipsCode ac_gte_op_cross_v3s4(AtomBuilder_R ab, Reg_(V3_S4) a, Reg_(V3_S4) b) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_gte_mv_to_cr_diag_v3s4(a), GteDelay_ /* RT diagonal: D1 = a.x, D2 = a.y, D3 = a.z */
|
||||||
|
mac_gte_ld_ir123_v3s4(b), GteDelay_ /* IR: second operand (b.xyz) */
|
||||||
|
gte_cmdw_cross, /* OP: MAC1/2/3 = a × b (S12.20) */
|
||||||
|
mac_gte_mv_from_mac123_v3s4(a), GteDelay_ /* Read MAC1/2/3 → a.xyz (overwrites source-A's load targets) */
|
||||||
|
mac_shift_aright_v3s4_self(a, 12), /* Right-shift MAC by 12 (S12.20 → S12.0 OuterProduct12) */
|
||||||
|
})
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||||
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
@@ -28,9 +64,9 @@ FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom
|
|||||||
|
|
||||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||||
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||||
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), LdSlot_ gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||||
})
|
})
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||||
@@ -38,7 +74,7 @@ I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v
|
|||||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||||
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, Reg r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||||
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||||
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||||
@@ -51,9 +87,7 @@ FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor)
|
|||||||
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||||
|
|
||||||
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
||||||
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
|
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs. */
|
||||||
* Stage 2 of normalize consumes these directly.
|
|
||||||
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
|
|
||||||
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
|
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
|
||||||
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
||||||
@@ -61,16 +95,13 @@ FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4
|
|||||||
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
||||||
})
|
})
|
||||||
|
|
||||||
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ───
|
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ─── */
|
||||||
* The SQR command always squares IR1/IR2/IR3 — those C2 registers are fixed.
|
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode delay_slot)
|
||||||
* The GPRs holding the source vector are caller-determined.
|
|
||||||
* Words: 5 (3 mtc2 + 1 nop hazard + 1 cmd). */
|
|
||||||
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode nop_slot)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
nop_slot, gte_cmdw_sqr,
|
delay_slot, gte_cmdw_sqr,
|
||||||
})
|
})
|
||||||
|
|
||||||
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
||||||
@@ -87,7 +118,7 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|||||||
gte_mv_to_data_r(r_sx, C2_IR1),
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
gte_mv_to_data_r(r_sy, C2_IR2),
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
gte_mv_to_data_r(r_sz, C2_IR3),
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
GteDelay_ nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
||||||
gte_cmdw_gpf,
|
gte_cmdw_gpf,
|
||||||
gte_mv_from_data_r(r_dx, C2_MAC1),
|
gte_mv_from_data_r(r_dx, C2_MAC1),
|
||||||
gte_mv_from_data_r(r_dy, C2_MAC2),
|
gte_mv_from_data_r(r_dy, C2_MAC2),
|
||||||
@@ -115,25 +146,21 @@ FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
|
|||||||
})
|
})
|
||||||
|
|
||||||
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
|
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
|
||||||
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32
|
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32 per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3 add).
|
||||||
* per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3
|
* Produces:
|
||||||
* add). Produces:
|
|
||||||
* r_shift ← LZCR rounded down to even (clear bit 0)
|
* r_shift ← LZCR rounded down to even (clear bit 0)
|
||||||
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
|
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
|
||||||
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
|
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
|
||||||
*
|
*
|
||||||
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division
|
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division is consistent — no 0.5 loss.
|
||||||
* is consistent — no 0.5 loss. The caller branches on LZCR < 24 to decide
|
* The caller branches on LZCR < 24 to decide left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
|
||||||
* left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
|
|
||||||
*
|
*
|
||||||
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller
|
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller must read it via mfc2 from C2_LZCR;
|
||||||
* must read it via mfc2 from C2_LZCR; there is no register choice at the
|
* there is no register choice at the hardware level. Only the GPR that holds the result is caller-determined. */
|
||||||
* hardware level. Only the GPR that holds the result is caller-determined. */
|
|
||||||
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
||||||
U4 r_shift,
|
U4 r_shift,
|
||||||
U4 r_mag_sq,
|
U4 r_mag_sq,
|
||||||
U4 r_mag_sq_copy
|
U4 r_mag_sq_copy)
|
||||||
)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
||||||
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
||||||
@@ -142,25 +169,6 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|||||||
shift_aright(r_mag_sq, r_mag_sq, 1),
|
shift_aright(r_mag_sq, r_mag_sq, 1),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
|
|
||||||
, Reg rd_v0, Reg rd_v1, Reg rd_v2
|
|
||||||
, Reg rs_v0, Reg rs_v1, Reg rs_v2
|
|
||||||
, Reg r_shift)
|
|
||||||
MipsAtomComp_Proc_(ab, {
|
|
||||||
shift_aright_var(rd_v0, rs_v0, r_shift),
|
|
||||||
shift_aright_var(rd_v1, rs_v1, r_shift),
|
|
||||||
shift_aright_var(rd_v2, rs_v2, r_shift),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab
|
|
||||||
, Reg rds_v0, Reg rds_v1, Reg rds_v2
|
|
||||||
, Reg r_shift)
|
|
||||||
MipsAtomComp_Proc_(ab, {
|
|
||||||
shift_aright_var(rds_v0, rds_v0, r_shift),
|
|
||||||
shift_aright_var(rds_v1, rds_v1, r_shift),
|
|
||||||
shift_aright_var(rds_v2, rds_v2, r_shift),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
||||||
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
|
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
|
||||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
||||||
@@ -170,31 +178,32 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||||
gte_mv_to_data_r(to_ir2, C2_IR2),
|
gte_mv_to_data_r(to_ir2, C2_IR2),
|
||||||
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
||||||
LdSlot_ nop_slot1,
|
GteDelay_ nop_slot1,
|
||||||
LdSlot_ nop_slot2,
|
GteDelay_ nop_slot2,
|
||||||
gte_cmdw_gpf,
|
gte_cmdw_gpf,
|
||||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
FI_ Slice_MipsCode ac_gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
||||||
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3)
|
||||||
)
|
|
||||||
MipsAtomComp_Proc_(ab, {
|
MipsAtomComp_Proc_(ab, {
|
||||||
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||||
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||||
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||||
})
|
})
|
||||||
|
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_mv_from_mac123_v3s4(AtomBuilder_R ab, Reg_(V3_S4) v) MipsAtomComp_ProcMap_(ab, mac_gte_mv_from_data_r_mac123(v.x, v.y, v.z))
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
#pragma region Atom Procs
|
#pragma region Atom Procs
|
||||||
|
|
||||||
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||||
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||||
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
|
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
||||||
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
|
||||||
*
|
*
|
||||||
* Data is identical to the libgte original (byte-for-byte verified).
|
* Data is identical to the libgte original (byte-for-byte verified).
|
||||||
*
|
*
|
||||||
@@ -229,7 +238,8 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
* and the load upper_halves of the table bracket the input range.
|
* and the load upper_halves of the table bracket the input range.
|
||||||
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||||
*
|
*
|
||||||
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
|
* Reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804).
|
||||||
|
* */
|
||||||
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||||
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||||
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||||
@@ -257,56 +267,41 @@ internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
|||||||
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||||
};
|
};
|
||||||
|
|
||||||
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
|
typedef Struct_(Binds_NormalizeV3S4) {
|
||||||
typedef Struct_(RegUse_normalize_v3s4_proc) {
|
U2 src_offset; /* offset of src V3_S4 within the BIOS scratchpad */
|
||||||
Reg scratch; // Scratch base carrier.
|
U2 dst_offset; /* offset of dst V3_S4 within the BIOS scratchpad */
|
||||||
|
};
|
||||||
|
typedef Struct_(RegUse_build_normalize_v3s4) {
|
||||||
|
Reg scratch; /* scratchpad base; loaded via load_word_imm below. */
|
||||||
Reg src_ptr;
|
Reg src_ptr;
|
||||||
Reg dst_ptr;
|
Reg dst_ptr;
|
||||||
Reg recip_est; // |v|² sum + shift-input + sqrtbl[index]
|
Reg recip_est; /* |v|² sum + shift-input + sqrtbl[index] */
|
||||||
Reg norm; Reg shift;
|
Reg norm; Reg shift;
|
||||||
Reg src_x;
|
Reg src_x;
|
||||||
union { Reg mac1_scratch; } t3;
|
union { Reg mac1_scratch, dst_offset; } t3;
|
||||||
union { Reg mac2_scratch; } t4;
|
union { Reg mac2_scratch; } t4;
|
||||||
union { Reg shift_count, btarget, lookup_addr, src_z; } t5;
|
union { Reg btarget, shift_count, lookup_addr, src_z, src_offset; } t5;
|
||||||
};
|
};
|
||||||
/* ─── Full normalize (all 4 stages inline) ───
|
/* ─── Full normalize (all 4 stages inline) ───
|
||||||
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav).
|
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav). */
|
||||||
*
|
internal MipsAtom* build_normalize_v3s4(AtomArena_R aa, RegUse_build_normalize_v3s4 r)
|
||||||
* Parameterized by caller-provided scratch base + src/dst offsets.
|
|
||||||
* The caller passes r_src_offset and r_dst_offset as compile-time constants
|
|
||||||
* (typically derived from O_ macros in the caller's struct schema, e.g., `O_(CallerBundleScratch, fwd)`).
|
|
||||||
*
|
|
||||||
* This design lets any caller (with a scratch base + struct schema) use `normalize_v3s4_proc`
|
|
||||||
* without putting magic offsets in the C-side bundle helper — the offsets come from O_ macros at the call site.
|
|
||||||
*
|
|
||||||
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
|
|
||||||
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
|
|
||||||
* r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4
|
|
||||||
* r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3)
|
|
||||||
* r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra
|
|
||||||
* r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y
|
|
||||||
* r_norm : |v|² sum (stage 2) → half-shift (stage 3) → 1/|v| (stage 4 IR0)
|
|
||||||
* r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav
|
|
||||||
* r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr)
|
|
||||||
*
|
|
||||||
* Atom_labels are srav_path / aligned_done
|
|
||||||
* (NOT namespaced — they're internal to this proc;
|
|
||||||
* the metaprogram's per-atom-name enum emission handles any collision across different atoms/files that share the same labels).
|
|
||||||
*
|
|
||||||
* Pool cost: 11 GPRs (well within the 9-10 caller-trash GPR budget when r_scratch is a wave-context carrier).
|
|
||||||
*
|
|
||||||
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
|
|
||||||
* Words: ~59 (matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
|
|
||||||
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
|
|
||||||
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
|
|
||||||
*/
|
|
||||||
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U2 src_offset, U2 dst_offset, RegUse_normalize_v3s4_proc r)
|
|
||||||
MipsAtom_Proc_(aa, {
|
MipsAtom_Proc_(aa, {
|
||||||
add_si(r.src_ptr, r.scratch, src_offset), /* r_src_ptr = &src */
|
/* Load scratch base via immediate (always Scratchpad_Loc = 0x1F800000 — the BIOS
|
||||||
|
* scratchpad, aliased by every consumer's ResolveLookAtScratch struct). */
|
||||||
|
mac_load_word_imm(r.scratch, Scratchpad_Loc),
|
||||||
|
/* Tape pop: src_offset, dst_offset = 4 bytes (packed into 1 U4: low16=src, high16=dst).
|
||||||
|
* Loads back-to-back fill each other's load-delay slots; the subsequent add_u
|
||||||
|
* (2 cycles after the matching load) sees a valid value. */
|
||||||
|
load_half(r.t5.src_offset, R_TapePtr, O_(Binds_NormalizeV3S4, src_offset)),
|
||||||
|
load_half(r.t3.dst_offset, R_TapePtr, O_(Binds_NormalizeV3S4, dst_offset)),
|
||||||
|
LdSlot_ add_u(r.src_ptr, r.scratch, r.t5.src_offset),
|
||||||
|
LdSlot_ add_u(r.dst_ptr, r.scratch, r.t3.dst_offset),
|
||||||
|
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_NormalizeV3S4)),
|
||||||
|
|
||||||
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
|
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
|
||||||
* r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */
|
* r.rt1_src_x holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below).
|
||||||
mac_load_v3s4(r.src_x, r.recip_est, r.t5.lookup_addr, r.src_ptr, 0),
|
* t5.src_offset/dst_offset are dead by here; t5 is reused for src.z in the mac_load_word_v3 below. */
|
||||||
|
mac_load_word_v3(r.src_x, r.recip_est, r.t5.src_z, r.src_ptr, 0),
|
||||||
|
|
||||||
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
||||||
LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
|
LdSlot_ mac_gte_sqr_v3s4(r.src_x, r.recip_est, r.t5.src_z, LdSlot_ nop),
|
||||||
@@ -315,8 +310,8 @@ MipsAtom_Proc_(aa, {
|
|||||||
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
|
mac_gte_mv_from_data_r_mac123(r.t3.mac1_scratch, r.t4.mac2_scratch, r.norm), LdSlot_ nop,
|
||||||
add_u_self( r.norm, r.t3.mac1_scratch),
|
add_u_self( r.norm, r.t3.mac1_scratch),
|
||||||
add_u_self( r.norm, r.t4.mac2_scratch),
|
add_u_self( r.norm, r.t4.mac2_scratch),
|
||||||
gte_mv_to_data_r( r.norm, C2_LZCS), LdSlot_ nop2,
|
gte_mv_to_data_r( r.norm, C2_LZCS), GteDelay_ nop2,
|
||||||
gte_mv_from_data_r(r.shift, C2_LZCR), LdSlot_ nop,
|
gte_mv_from_data_r(r.shift, C2_LZCR), GteDelay_ nop,
|
||||||
|
|
||||||
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
|
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
|
||||||
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
|
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
|
||||||
@@ -350,17 +345,42 @@ MipsAtom_Proc_(aa, {
|
|||||||
r.recip_est,
|
r.recip_est,
|
||||||
r.t5.src_z, /* IR3 = src.z (reloaded) */
|
r.t5.src_z, /* IR3 = src.z (reloaded) */
|
||||||
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
|
r.t4.mac2_scratch, r.recip_est, r.t5.src_z,
|
||||||
LdSlot_ add_si(r.dst_ptr, r.scratch, dst_offset), // pre-laoding destination to register here.
|
GteDelay_ nop,
|
||||||
LdSlot_ nop
|
GteDelay_ nop
|
||||||
),
|
),
|
||||||
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
|
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
|
||||||
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
|
mac_shift_aright_var_v3_self(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.shift),
|
||||||
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
|
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
|
||||||
mac_store_v3s4(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
|
mac_store_word_v3(r.t4.mac2_scratch, r.recip_est, r.t5.src_z, r.dst_ptr, 0),
|
||||||
|
|
||||||
mac_yield()
|
mac_yield()
|
||||||
})
|
})
|
||||||
|
|
||||||
|
/* ─── GTE OP cross product (a × b → out) ───
|
||||||
|
* Generalized V3_S4 cross product via GTE OP (OuterProduct12 libpsyx convention).
|
||||||
|
* The >> 12 shift converts S12.20 → S12.0 OuterProduct12. */
|
||||||
|
typedef Struct_(Binds_gte_cross_v3s4) { V3_S4* src_a; V3_S4* src_b; V3_S4* out; };
|
||||||
|
typedef Struct_(RegUse_gte_cross_v3s4) {
|
||||||
|
Reg_(V3_S4) a;
|
||||||
|
Reg_(V3_S4) b;
|
||||||
|
union { Reg out, t0; } x;
|
||||||
|
union { Reg src_a, t1, rt11; } y;
|
||||||
|
union { Reg src_b, t2, rt22; } z;
|
||||||
|
};
|
||||||
|
internal MipsAtom* gte_cross_v3s4(AtomArena_R aa, RegUse_gte_cross_v3s4 r)
|
||||||
|
atom_info(atom_bind(Binds_gte_cross_v3s4)) MipsAtom_Proc_(aa, {
|
||||||
|
load_word(r.y.src_a, R_TapePtr, O_(Binds_gte_cross_v3s4,src_a)),
|
||||||
|
load_word(r.z.src_b, R_TapePtr, O_(Binds_gte_cross_v3s4,src_b)),
|
||||||
|
load_word(r.x.out, R_TapePtr, O_(Binds_gte_cross_v3s4,out)),
|
||||||
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_gte_cross_v3s4)),
|
||||||
|
|
||||||
|
mac_load_v3s4(r.a, r.y.src_a, 0), LdSlot_
|
||||||
|
mac_load_v3s4(r.b, r.z.src_b, 0), LdSlot_
|
||||||
|
mac_gte_op_cross_v3s4(r.a, r.b), /* RT diagonal + IR + OP + MAC read + shift */
|
||||||
|
mac_store_v3s4(r.a, r.x.out, 0),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
#pragma endregion Atom Procs
|
#pragma endregion Atom Procs
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
#pragma region Baked Atoms
|
||||||
@@ -376,12 +396,22 @@ internal MipsAtom_(set_gte_mt3s2s4) atom_info(
|
|||||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
||||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||||
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
load_word(R_T0, R_T3, 0),
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
load_word(R_T1, R_T3, 4),
|
||||||
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11),
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||||
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
load_word(R_T0, R_T3, 8),
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
load_word(R_T1, R_T3, 12),
|
||||||
|
load_word(R_T2, R_T3, 16),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13),
|
||||||
|
gte_mv_to_ctrl_r(R_T1, gte_cr_RT21),
|
||||||
|
gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||||
|
load_word(R_T0, R_T3, 20),
|
||||||
|
load_word(R_T1, R_T3, 24),
|
||||||
|
load_word(R_T2, R_T3, 28),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX),
|
||||||
|
gte_mv_to_ctrl_r(R_T1, gte_cr_TRY),
|
||||||
|
gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||||
mac_yield()
|
mac_yield()
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
+38
-51
@@ -16,9 +16,6 @@
|
|||||||
* gte_mv_to_data_r (gte + mv + to + data + register)
|
* gte_mv_to_data_r (gte + mv + to + data + register)
|
||||||
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
* gte_lw_v0_xy(base) (gte + lw + v0 + xy)
|
||||||
* load_upper_i (load-upper + immediate, unique verb)
|
* load_upper_i (load-upper + immediate, unique verb)
|
||||||
*
|
|
||||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, gte_lwc2, gte_swc2, etc.) are NOT in this header.
|
|
||||||
* They are in the opt-in `gte_vendor_sym.h` for users who prefer the textbook MIPS assembly mnemonics.
|
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
@@ -33,7 +30,7 @@
|
|||||||
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
* gte.h — Geometry Transformation Engine (COP2) for the PS1
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* Hand-rolled DSL for emitting GTE/MIPS instruction words as raw `.word` constants from C.
|
* Hand-rolled DSL for emitting GTE/MIPS instruction words from C.
|
||||||
* No GCC inline-assembly string syntax in the code body.
|
* No GCC inline-assembly string syntax in the code body.
|
||||||
*
|
*
|
||||||
* STYLE NOTES
|
* STYLE NOTES
|
||||||
@@ -101,20 +98,20 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for GTE Data Registers */
|
/* Semantic Aliases for GTE Data Registers */
|
||||||
enum {
|
enum {
|
||||||
gte_in_v0_xy = C2_VXY0, /* Input Vector 0 (X, Y) */
|
C2_InV0_XY = C2_VXY0, /* Input Vector 0 (X, Y) */
|
||||||
gte_in_v0_z = C2_VZ0, /* Input Vector 0 (Z) */
|
C2_InV0_Z = C2_VZ0, /* Input Vector 0 (Z) */
|
||||||
gte_in_v1_xy = C2_VXY1, /* Input Vector 1 (X, Y) */
|
C2_InV1_XY = C2_VXY1, /* Input Vector 1 (X, Y) */
|
||||||
gte_in_v1_z = C2_VZ1, /* Input Vector 1 (Z) */
|
C2_InV1_Z = C2_VZ1, /* Input Vector 1 (Z) */
|
||||||
gte_in_v2_xy = C2_VXY2, /* Input Vector 2 (X, Y) */
|
C2_InV2_XY = C2_VXY2, /* Input Vector 2 (X, Y) */
|
||||||
gte_in_v2_z = C2_VZ2, /* Input Vector 2 (Z) */
|
C2_InV2_Z = C2_VZ2, /* Input Vector 2 (Z) */
|
||||||
gte_in_rgb = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
C2_In_RGB = C2_RGB, /* Input Color (R, G, B, MipsCode) */
|
||||||
gte_out_scr_xy0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
C2_OutSrc_XY0 = C2_SXY0, /* Output Screen Coord 0 (X, Y) */
|
||||||
gte_out_scr_xy1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
C2_OutSrc_XY1 = C2_SXY1, /* Output Screen Coord 1 (X, Y) */
|
||||||
gte_out_scr_xy2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
C2_OutSrc_XY2 = C2_SXY2, /* Output Screen Coord 2 (X, Y) */
|
||||||
gte_out_depth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
C2_OutDepth = C2_OTZ, /* Output Ordering Table Z (Depth) */
|
||||||
gte_math_accum0 = C2_MAC0, /* Math Accumulator 0 */
|
C2_MathAccu0 = C2_MAC0, /* Math Accumulator 0 */
|
||||||
gte_math_accum1 = C2_MAC1, /* Math Accumulator 1 */
|
C2_MathAccu1 = C2_MAC1, /* Math Accumulator 1 */
|
||||||
gte_math_accum2 = C2_MAC2, /* Math Accumulator 2 */
|
C2_MathAccu2 = C2_MAC2, /* Math Accumulator 2 */
|
||||||
};
|
};
|
||||||
|
|
||||||
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
/* --- GTE Command Semantics (The Bitfield Meanings) ---
|
||||||
@@ -191,27 +188,22 @@ enum {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
||||||
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon:
|
* Three pairs of aliases map to the C2 control-register slot:
|
||||||
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
||||||
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
||||||
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
||||||
* Cross-alias writes inside one atom body, or across the wave-context boundary,
|
* Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
||||||
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes
|
* The metaprogram's check_gte_cr_alias_writes (CHECK_RULES row) warns about each pair per source.
|
||||||
* (CHECK_RULES row) warns about each pair per source. See
|
* See psx-spx docs/gte_reference.md §"Control-register alias table" for the silicon rationale and the libgte outer-product convention.
|
||||||
* docs/gte_reference.md §"Control-register alias table" for the silicon
|
|
||||||
* rationale and the libgte outer-product convention.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
||||||
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
||||||
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
||||||
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
||||||
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product
|
* OP and MVMVA read D1/D2/D3 from these packed slots.
|
||||||
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then
|
* The libgte outer-product convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then C2[4] in sequence;
|
||||||
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent
|
* the SECOND write's low half is RT22, not RT13.
|
||||||
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
|
|
||||||
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
|
|
||||||
* for the canonical write pattern.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||||
@@ -300,8 +292,7 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
// #define gte_mv_from_data_r(rt, rd) enc_gte_tx(cop_mf, (rt), (rd)) /* Move GTE Control Register (rd) to GPR (rt) */
|
||||||
|
|
||||||
/* GTE Data vs Control Register Transfers
|
/* GTE Data vs Control Register Transfers
|
||||||
*
|
* Each macro emits a single instruction for one of MFC2/CFC2/MTC2/CTC2.
|
||||||
* Each macro emits a single .word constant for one of MFC2/CFC2/MTC2/CTC2.
|
|
||||||
*
|
*
|
||||||
* `rd` is the C2 register index in the file the sub-opcode names:
|
* `rd` is the C2 register index in the file the sub-opcode names:
|
||||||
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
* gte_mv_from_data_r / gte_mv_to_data_r → C2 data register file
|
||||||
@@ -316,14 +307,14 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
#define gte_mv_from_ctrl_r(rt, rd) enc_gte_tx(sub_cfc2, (rt), (rd)) /* Copy From ctrl reg */
|
||||||
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
#define gte_mv_to_data_r(rt, rd) enc_gte_tx(sub_mtc2, (rt), (rd)) /* Move To data reg */
|
||||||
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
#define gte_mv_to_ctrl_r(rt, rd) enc_gte_tx(sub_ctc2, (rt), (rd)) /* Copy To ctrl reg */
|
||||||
|
#define GteDelay_ // Annotate an instruction as filling a CPU <-> GTE DMA delay slot/s
|
||||||
|
|
||||||
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
/* COP2 Data Load (lwc2): `lwc2 rt, off(rs)`
|
||||||
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
* Layout: [op_lwc2:6][rs:5][rt:5][imm:16]
|
||||||
* - rs: GPR base address
|
* - rs: GPR base address
|
||||||
* - rt: COP2 data register index (0..31)
|
* - rt: COP2 data register index (0..31)
|
||||||
* - imm: signed 16-bit offset
|
* - imm: signed 16-bit offset
|
||||||
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked
|
* NOTE: When `rs` is a runtime register, the encoding cannot be pre-baked into a .word — use the string-style `gte_load_v0` macro below instead. */
|
||||||
* into a .word — use the string-style `gte_load_v0` macro below instead. */
|
|
||||||
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
#define enc_gte_lw(rt, base, off) enc_i(op_lwc2, (base), (rt), (off))
|
||||||
/* Store Word */
|
/* Store Word */
|
||||||
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
#define enc_gte_sw(rt, base, off) enc_i(op_swc2, (base), (rt), (off))
|
||||||
@@ -332,8 +323,7 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
* `swc2` is redundant when we're already inside the `gte_` namespace.
|
||||||
* gte_lw rt, base, off → lwc2 rt, off(base)
|
* gte_lw rt, base, off → lwc2 rt, off(base)
|
||||||
* gte_sw rt, base, off → swc2 rt, off(base)
|
* gte_sw rt, base, off → swc2 rt, off(base)
|
||||||
* For the typical user-facing vector-level load (xy + z as two instructions),
|
* For the typical user-facing vector-level load (xy + z as two instructions), use the higher-level `gte_load_vN` macros below. */
|
||||||
* use the higher-level `gte_load_vN` macros below. */
|
|
||||||
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
#define gte_lw(rt, base, off) enc_gte_lw(rt, base, off)
|
||||||
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
#define gte_sw(rt, base, off) enc_gte_sw(rt, base, off)
|
||||||
|
|
||||||
@@ -408,10 +398,11 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
#define gte_cmdw_rtpt (gte_cmd_base | enc_gte_cmd(gte_cmd_rtpt ) | gte_cmdw_psyq_compat)
|
||||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||||
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
|
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- PSY-Q terminology */
|
||||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
|
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology. */
|
||||||
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
#define gte_cmdw_cross gte_cmdw_op /* "cross product" -- geometric-algebra terminology.
|
||||||
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
* RGA(Lengyel): The GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
||||||
|
* The wedge alias is a 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
||||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
||||||
@@ -448,9 +439,8 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
||||||
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's
|
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's ApplyMatrixLV when the GTE pipeline expects R*pos >> 12.
|
||||||
* ApplyMatrixLV when the GTE pipeline expects R*pos >> 12. The shift produces
|
* The shift produces values like (-270, 710, 1713) which match the C11 reference path. */
|
||||||
* values like (-270, 710, 1713) which match the C11 reference path. */
|
|
||||||
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
||||||
|
|
||||||
/* SQR / GPF cosmetic-bits compat helpers.
|
/* SQR / GPF cosmetic-bits compat helpers.
|
||||||
@@ -483,10 +473,8 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
||||||
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
||||||
|
|
||||||
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31)
|
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31) down to even.
|
||||||
* down to even. The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1;
|
* The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1; clearing bit 0 ensures the subtraction result is always odd, so the >> 1 division is consistent (no 0.5 loss). */
|
||||||
* clearing bit 0 ensures the subtraction result is always odd,
|
|
||||||
* so the >> 1 division is consistent (no 0.5 loss). */
|
|
||||||
enum {
|
enum {
|
||||||
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
|
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
|
||||||
};
|
};
|
||||||
@@ -593,8 +581,8 @@ enum {
|
|||||||
|
|
||||||
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
/* gte_load_v0v1v2(p0, p1, p2, b0, b1, b2) — prelude to gte_cmd_rtpt.
|
||||||
*
|
*
|
||||||
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register,
|
* Loads all three GTE input vectors (6 words) from three separate pointers, one per GTE vector register, each loaded from its own base GPR.
|
||||||
* each loaded from its own base GPR. Caller must bind each `pN` to `bN` via a register variable.
|
* Caller must bind each `pN` to `bN` via a register variable.
|
||||||
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
* register V3_S2* p0 rgcc(R_T4) = verts[0].ptr; // → __asm__("$12")
|
||||||
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
* register V3_S2* p1 rgcc(R_T5) = verts[1].ptr; // → __asm__("$13")
|
||||||
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
* register V3_S2* p2 rgcc(R_T6) = verts[2].ptr; // → __asm__("$14")
|
||||||
@@ -688,8 +676,7 @@ enum {
|
|||||||
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
* Loads the 3x3 rotation matrix at `r0` into the GTE's rotation-matrix control registers (RT11..RT22, indices 0..4) via ctc2.
|
||||||
*
|
*
|
||||||
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
* Memory layout at r0: five contiguous 32-bit words (offsets 0..16), each holding two packed 16-bit matrix elements.
|
||||||
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as
|
* The first 1.5 rows of a standard PSX SDK MATRIX struct (where each row is laid out as [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
||||||
* [RT_xx, RT_xy] | [RT_xz, pad] | ...).
|
|
||||||
*
|
*
|
||||||
* Generated MIPS (mirrors the source macro):
|
* Generated MIPS (mirrors the source macro):
|
||||||
* lw $12, 0( %0 ) ; word 0
|
* lw $12, 0( %0 ) ; word 0
|
||||||
|
|||||||
+151
-88
@@ -66,51 +66,68 @@
|
|||||||
* */
|
* */
|
||||||
/* Register Allocation Info */
|
/* Register Allocation Info */
|
||||||
enum {
|
enum {
|
||||||
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
|
R_ScratchBase = R_SP atom_reg, /* Scratchpad base address (host frame top) */
|
||||||
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
|
R_AtomJmp = R_FP atom_reg, /* Next atom target (yield handshake scratch) */
|
||||||
|
R_TapePtr = R_RA atom_reg, /* The Instruction Stream Pointer */
|
||||||
/* Stringification codes for the GCC inline assembler clobber lists. */
|
/* Stringification codes for the GCC inline assembler clobber lists. */
|
||||||
#define R_AtomJmp_Code R_T8_Code
|
#define R_ScratchBase_Code R_SP_Code
|
||||||
#define R_TapePtr_Code R_T9_Code
|
#define R_AtomJmp_Code R_FP_Code
|
||||||
|
#define R_TapePtr_Code R_RA_Code
|
||||||
|
|
||||||
// R_InCursor = R_T4,
|
// R_InCursor = R_T4,
|
||||||
// #define R_InCursor_Code R_T4_Code
|
// #define R_InCursor_Code R_T4_Code
|
||||||
|
|
||||||
// Reserved Registers (Callee-saved):
|
// Reserved Registers (Callee-saved across the host ABI transition):
|
||||||
// - R_T9: Holds the Tape Ptr which we need to increment
|
// - R_SP: Holds the scratchpad base while tape code executes.
|
||||||
// If we hit a wall with register allocations we can clobber V0 & V1 (return values), defering as opt-in by user.
|
// - R_FP: Holds the next atom target.
|
||||||
// - R_RA: Not sure??
|
// - R_RA: Holds the tape cursor.
|
||||||
// Needed by ac_yield but can be used as atom scratch:
|
// All atom-body allocations must stay out of these.
|
||||||
// - R_T8: Will be used as the atom jump register.
|
// Atom bodies may freely use R2-R25.
|
||||||
|
|
||||||
// All allocatable registers for mips atoms:
|
// All allocatable registers for atom bodies (R2-R25, 24 registers):
|
||||||
R_TScratchVolatile = R_AT, // This one is reserved for psuedo instructions, but you can technically use it.
|
|
||||||
R_TScratch0 = R_T0,
|
R_PsuedoVolatile = R_AT, // Assembler temporary; never allocate.
|
||||||
R_TScratch1 = R_T1,
|
|
||||||
R_TScratch2 = R_T2,
|
// Atom Allocation Pool
|
||||||
R_TScratch3 = R_T3,
|
R_Atom0 = R_T0,
|
||||||
R_TScratch4 = R_T4,
|
R_Atom1 = R_T1,
|
||||||
R_TScratch5 = R_T5,
|
R_Atom2 = R_T2,
|
||||||
R_TScratch6 = R_T6,
|
R_Atom3 = R_T3,
|
||||||
R_TScratch7 = R_T7,
|
R_Atom4 = R_T4,
|
||||||
R_TScratch8 = R_T8,
|
R_Atom5 = R_T5,
|
||||||
R_TScratch10 = R_V0, // Tend to be used with gte DMAs
|
R_Atom6 = R_T6,
|
||||||
R_TScratch11 = R_V1, // Tend to be used with gte DMAs
|
R_Atom7 = R_T7,
|
||||||
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
|
R_Atom8 = R_T8,
|
||||||
// A 0-2
|
R_Atom9 = R_T9,
|
||||||
// S 0-7
|
R_Atom10 = R_V0, // Tend to be used with gte DMAs
|
||||||
|
R_Atom11 = R_V1, // Tend to be used with gte DMAs
|
||||||
|
R_Atom12 = R_A0,
|
||||||
|
R_Atom13 = R_A1,
|
||||||
|
R_Atom14 = R_A2,
|
||||||
|
R_Atom15 = R_A3,
|
||||||
|
R_Atom16 = R_S0,
|
||||||
|
R_Atom17 = R_S1,
|
||||||
|
R_Atom18 = R_S2,
|
||||||
|
R_Atom19 = R_S3,
|
||||||
|
R_Atom20 = R_S4,
|
||||||
|
R_Atom21 = R_S5,
|
||||||
|
R_Atom22 = R_S6,
|
||||||
|
R_Atom23 = R_S7,
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
||||||
|
#define Reg_(type) tmpl(Reg,type) // Just a way to template register allocations of C-struct types.
|
||||||
|
|
||||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||||
typedef Slice_(MipsCode);
|
typedef Slice_(MipsCode);
|
||||||
|
|
||||||
typedef U4 const MipsAtom;
|
typedef U4 const MipsAtom; // Underlying type to a mips atom defnition
|
||||||
typedef Slice_(MipsAtom);
|
typedef Slice_(MipsAtom);
|
||||||
|
|
||||||
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
||||||
// MipsAtom* <identifier>[...];
|
// MipsAtom* <identifier>[...];
|
||||||
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
||||||
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the blow returns 'MipsAtom'.
|
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the proc returns 'MipsAtom'.
|
||||||
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
||||||
|
|
||||||
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
||||||
@@ -143,57 +160,78 @@ typedef Slice_(MipsAtom);
|
|||||||
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
||||||
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
|
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
|
||||||
|
// Used for trivial mappings from one atom component proc to the command of a more baser (meant for type-mapping)
|
||||||
|
#define MipsAtomComp_ProcMap_(ab, base_command) atom_dbg_skip MipsAtomComp_Proc_(ab, {base_command })
|
||||||
|
|
||||||
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||||
Files containing only atoms and atom components.
|
Files containing only atoms and atom components.
|
||||||
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||||
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
||||||
The constant is in `.rodata` so the linker may eliminate it.
|
The constant is in `.rodata` so the linker may eliminate it. */
|
||||||
Two-level concat + `__LINE__` suffix makes the identifier unique per call site
|
|
||||||
(identifier embeds the source line, so duplicates across `#include`d files don't collide). */
|
|
||||||
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||||
|
|
||||||
typedef Slice_MipsAtom Tape;
|
typedef Slice_MipsAtom Tape;
|
||||||
|
|
||||||
/* The 'Exit' Atom */
|
typedef Struct_(TapeHostFrame) {
|
||||||
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(R_RA), nop };
|
U4 s0;
|
||||||
|
U4 s1;
|
||||||
|
U4 s2;
|
||||||
|
U4 s3;
|
||||||
|
U4 s4;
|
||||||
|
U4 s5;
|
||||||
|
U4 s6;
|
||||||
|
U4 s7;
|
||||||
|
U4 fp;
|
||||||
|
U4 sp;
|
||||||
|
U4 ra;
|
||||||
|
};
|
||||||
|
|
||||||
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
|
enum {
|
||||||
|
TapeHostFrame_Loc = Scratchpad_End - S_(TapeHostFrame),
|
||||||
|
TapeScratch_Len = TapeHostFrame_Loc - Scratchpad_Loc,
|
||||||
|
};
|
||||||
|
static_assert(S_(TapeHostFrame) == 11 * S_(U4));
|
||||||
|
static_assert(TapeHostFrame_Loc == 0x1F8003D4);
|
||||||
|
|
||||||
/* Tape Runner (Default) */
|
atom_dbg_skip MipsAtom_(tape_enter) {
|
||||||
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
||||||
asm_words(
|
store_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
||||||
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
store_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
||||||
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
|
store_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
||||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
store_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
||||||
, nop /* Branch delay slot */
|
store_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
||||||
)
|
store_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
||||||
asm_rpins, r_use(tape_ptr)
|
store_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
||||||
asm_clobber:
|
store_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
||||||
rlit(R_AT),
|
store_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
||||||
rlit(R_V0), rlit(R_V1), // We clobber these for GTE ACs (that don't expose register selection, might expose them in the future...)
|
store_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
||||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
store_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
||||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
add_ui(R_TapePtr, R_A0, 0),
|
||||||
clb_mem_drain
|
load_upper_i(R_ScratchBase, u4_hi(Scratchpad_Loc)),
|
||||||
); }
|
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||||
|
add_ui_self( R_TapePtr, S_(MipsAtom)),
|
||||||
|
jump_reg(R_AtomJmp), BdSlot_ nop,
|
||||||
|
};
|
||||||
|
|
||||||
/* Tape Runner (Static and Arg Clobbers) */
|
atom_dbg_skip MipsAtom_(tape_exit) {
|
||||||
FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
mac_load_word_imm(R_V0, u4_(TapeHostFrame_Loc)),
|
||||||
asm_words(
|
load_word(R_S0, R_V0, O_(TapeHostFrame,s0)),
|
||||||
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
load_word(R_S1, R_V0, O_(TapeHostFrame,s1)),
|
||||||
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
|
load_word(R_S2, R_V0, O_(TapeHostFrame,s2)),
|
||||||
, call_reg( R_AtomJmp) /* jalr $t9 */
|
load_word(R_S3, R_V0, O_(TapeHostFrame,s3)),
|
||||||
, nop /* Branch delay slot */
|
load_word(R_S4, R_V0, O_(TapeHostFrame,s4)),
|
||||||
)
|
load_word(R_S5, R_V0, O_(TapeHostFrame,s5)),
|
||||||
asm_rpins, r_use(tape_ptr)
|
load_word(R_S6, R_V0, O_(TapeHostFrame,s6)),
|
||||||
asm_clobber:
|
load_word(R_S7, R_V0, O_(TapeHostFrame,s7)),
|
||||||
rlit(R_AT),
|
load_word(R_RA, R_V0, O_(TapeHostFrame,ra)),
|
||||||
rlit(R_V0), rlit(R_V1), rlit(R_A0), rlit(R_A1), rlit(R_A2),
|
load_word(R_FP, R_V0, O_(TapeHostFrame,fp)),
|
||||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
load_word(R_SP, R_V0, O_(TapeHostFrame,sp)),
|
||||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
jump_reg(R_RA), BdSlot_ nop,
|
||||||
rlit(R_S0), rlit(R_S1), rlit(R_S2), rlit(R_S3), rlit(R_S4),
|
};
|
||||||
rlit(R_S5), rlit(R_S6), rlit(R_S7),
|
|
||||||
clb_mem_drain
|
typedef void Proc_(TapeEntryFn)(MipsAtom* tape_ptr);
|
||||||
); }
|
|
||||||
|
FI_ void tape_run(Tape tape) { C_(TapeEntryFn*, tape_enter)(tape.ptr); }
|
||||||
|
|
||||||
// Procedural authoring of tapes:
|
// Procedural authoring of tapes:
|
||||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||||
@@ -223,15 +261,11 @@ FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_
|
|||||||
* ---------------------------------------------------------------------------*/
|
* ---------------------------------------------------------------------------*/
|
||||||
|
|
||||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
|
|
||||||
// - mac_yield_load() + mac_yield_tail():
|
|
||||||
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
|
|
||||||
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
|
|
||||||
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
atom_dbg_skip MipsAtomComp_(ac_yield) {
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0),
|
load_word(R_AtomJmp, R_TapePtr, 0),
|
||||||
add_ui_self( R_TapePtr, S_(MipsCode)),
|
add_ui_self( R_TapePtr, S_(MipsCode)),
|
||||||
jump_reg( R_AtomJmp), nop,
|
jump_reg( R_AtomJmp), BdSlot_ nop,
|
||||||
};
|
};
|
||||||
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
||||||
@@ -240,16 +274,13 @@ atom_dbg_skip MipsAtomComp_(ac_yield_load) {
|
|||||||
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
||||||
add_ui_self(R_TapePtr, S_(MipsCode)),
|
add_ui_self(R_TapePtr, S_(MipsCode)),
|
||||||
jump_reg( R_AtomJmp), nop,
|
jump_reg( R_AtomJmp), BdSlot_ nop,
|
||||||
};
|
};
|
||||||
|
|
||||||
#pragma endregion Macro Atom Components
|
#pragma endregion Macro Atom Components
|
||||||
|
|
||||||
#pragma region Atom Builder
|
#pragma region Atom Builder
|
||||||
// This helps with runtime procedural authoring of mips atoms.
|
// This helps with runtime procedural authoring of mips atoms.
|
||||||
|
|
||||||
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
|
||||||
|
|
||||||
// FArena Related
|
|
||||||
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||||
|
|
||||||
// Usual way to resolve an atom after the bulder is done.
|
// Usual way to resolve an atom after the bulder is done.
|
||||||
@@ -270,7 +301,6 @@ FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, a
|
|||||||
|
|
||||||
#pragma region Atom Arena
|
#pragma region Atom Arena
|
||||||
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
||||||
|
|
||||||
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
||||||
|
|
||||||
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
||||||
@@ -295,13 +325,24 @@ FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
|||||||
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
|
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
|
||||||
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
|
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
|
||||||
|
|
||||||
/* ABI + tape reserves that are never handed out by alloc. */
|
/* ABI reserves that are never handed out by alloc.
|
||||||
|
* R_AT is the assembler temporary (per the MIPS O32 ABI).
|
||||||
|
* R_K0/K1 are kernel reserves.
|
||||||
|
* R_GP stays the host global pointer.
|
||||||
|
* R_SP/R_FP/R_RA are tape runtime carriers between tape_enter and tape_exit. */
|
||||||
U4 const regfile_abi_mask =
|
U4 const regfile_abi_mask =
|
||||||
(1u << R_0) | (1u << R_AT) |
|
(1u << R_0) | (1u << R_AT) |
|
||||||
(1u << R_K0) | (1u << R_K1) |
|
(1u << R_K0) | (1u << R_K1) |
|
||||||
(1u << R_GP) | (1u << R_SP) |
|
(1u << R_GP) | (1u << R_SP) |
|
||||||
(1u << R_FP) | (1u << R_RA) |
|
(1u << R_FP) | (1u << R_RA);
|
||||||
(1u << R_T8) | (1u << R_T9); /* AtomJmp + TapePtr */
|
|
||||||
|
internal Reg const regfile_alloc_order[] = {
|
||||||
|
R_V0, R_V1,
|
||||||
|
R_A0, R_A1, R_A2, R_A3,
|
||||||
|
R_T0, R_T1, R_T2, R_T3, R_T4, R_T5, R_T6, R_T7,
|
||||||
|
R_S0, R_S1, R_S2, R_S3, R_S4, R_S5, R_S6, R_S7,
|
||||||
|
R_T8, R_T9,
|
||||||
|
};
|
||||||
|
|
||||||
typedef Struct_(RegFile) {
|
typedef Struct_(RegFile) {
|
||||||
A2_U2 GPR;
|
A2_U2 GPR;
|
||||||
@@ -336,13 +377,16 @@ FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
|
|||||||
}
|
}
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
/* regfile_alloc picks the next free GPR from regfile_alloc_order.
|
||||||
|
* The table is the first-fit allocation order: T0..T7, V0..V1, A0..A3,
|
||||||
|
* S0..S7, T8..T9. The 24 entries leave room for the tape program to use
|
||||||
|
* any of them while R0, R1, R26-R31 remain reserved. */
|
||||||
I_ Reg regfile_alloc(RegFile_R rf) {
|
I_ Reg regfile_alloc(RegFile_R rf) {
|
||||||
U2 allocated = 0;
|
Reg allocated = 0;
|
||||||
for index_iter(Reg, r_id, R_T0, <=, R_T7) {
|
for index_iter(U4, r_id, R_V0, <, R_T9) {
|
||||||
allocated = regfile__alloc_helper(rf->GPR, r_id); Jmp_nZero_(allocated,resolved);
|
allocated = regfile__alloc_helper(rf->GPR, r_id);
|
||||||
|
Jmp_nZero_(allocated,resolved);
|
||||||
}
|
}
|
||||||
allocated = regfile__alloc_helper(rf->GPR, R_V0); Jmp_nZero_(allocated,resolved);
|
|
||||||
allocated = regfile__alloc_helper(rf->GPR, R_V1);
|
|
||||||
assert(allocated != 0);
|
assert(allocated != 0);
|
||||||
resolved: return allocated;
|
resolved: return allocated;
|
||||||
}
|
}
|
||||||
@@ -371,13 +415,32 @@ FI_ void regfile_reset(RegFile_R rf) {
|
|||||||
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||||
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||||
}
|
}
|
||||||
FI_ void regfile_reset_mask(RegFile_R rf, U4 mask) {
|
FI_ void regfile_reset_to_mask(RegFile_R rf, U4 mask) {
|
||||||
rf->GPR[0] = u4_lo(mask);
|
rf->GPR[0] = u4_lo(mask);
|
||||||
rf->GPR[1] = u4_hi(mask);
|
rf->GPR[1] = u4_hi(mask);
|
||||||
}
|
}
|
||||||
#pragma endregion RegFileArena (Register File Allocator)
|
#pragma endregion RegFileArena (Register File Allocator)
|
||||||
|
|
||||||
#pragma region Mips Atom Procs
|
#pragma region Mips Atom Procs
|
||||||
|
/* RegUse structs are a convention to organize register allocations for a mips atom procedure.
|
||||||
|
Unlike the usual enum-based declarations, they provide a namespaced scope and have view types via union declarations. */
|
||||||
|
#define RegUse_(proc_name) (tmpl(RegUse,proc_name))
|
||||||
|
|
||||||
|
typedef Struct_(RegUse_example_atom_proc) {
|
||||||
|
Reg const ro_register; // Scratch base carrier.
|
||||||
|
Reg usual_modifiable;
|
||||||
|
union { Reg view_1, view_2, view_3; } t1;
|
||||||
|
};
|
||||||
|
internal MipsAtom* example_atom_proc(AtomArena_R aa, U2 offset, RegUse_example_atom_proc r)
|
||||||
|
MipsAtom_Proc_(aa, {
|
||||||
|
add_si(r.usual_modifiable, r.ro_register, offset),
|
||||||
|
or_u(r.t1.view_1, r.ro_register, 0),
|
||||||
|
branch_lt_zero(r.t1.view_1, atom_offset(example_atom_proc, skip)), BdSlot_ nop,
|
||||||
|
li_s(r.t1.view_2, 100),
|
||||||
|
atom_label(skip)
|
||||||
|
add_si(r.t1.view_3, r.usual_modifiable, 10),
|
||||||
|
mac_yield(),
|
||||||
|
})
|
||||||
|
|
||||||
#pragma endregion Mips Atom Procs
|
#pragma endregion Mips Atom Procs
|
||||||
|
|
||||||
|
|||||||
@@ -1,55 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
# include "gen/macs.h"
|
|
||||||
# include "gen/offsets.h"
|
|
||||||
# include "math.h"
|
|
||||||
# include "lottes_tape.h"
|
|
||||||
#endif
|
|
||||||
|
|
||||||
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
|
||||||
|
|
||||||
#pragma region MACs (Mips Atom Component)
|
|
||||||
|
|
||||||
// FI_ Slice_MipsCode ac_load_imm
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
|
|
||||||
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
|
||||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
|
|
||||||
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
|
|
||||||
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
|
|
||||||
})
|
|
||||||
// TODO(Ed): we could generate these mappings properly..
|
|
||||||
#define ac_load_p3s4 ac_load_v3s4
|
|
||||||
#define mac_load_p3s4 mac_load_v3s4
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_word(rt_x, base, offset + O_(V3_S4,x)),
|
|
||||||
store_word(rt_y, base, offset + O_(V3_S4,y)),
|
|
||||||
store_word(rt_z, base, offset + O_(V3_S4,z)),
|
|
||||||
})
|
|
||||||
// TODO(Ed): we could generate these mappings properly..
|
|
||||||
#define ac_store_p3s4 ac_store_v3s4
|
|
||||||
#define mac_store_p3s4 mac_store_v3s4
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
sub_s(rds_x, rds_x, rt_x),
|
|
||||||
sub_s(rds_y, rds_y, rt_y),
|
|
||||||
sub_s(rds_z, rds_z, rt_z),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
|
||||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
|
||||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
|
||||||
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
|
||||||
})
|
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Component)
|
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "math.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||||
|
|
||||||
|
#define v3s4_R_0() ((Reg_(V3_S4)){R_0,R_0,R_0})
|
||||||
|
|
||||||
|
typedef Struct_(Reg_V3_S2) { Reg x, y, z; };
|
||||||
|
typedef Struct_(Reg_V3_S4) { Reg x, y, z; }; // Register allocation of a V3_S4
|
||||||
|
typedef Struct_(Reg_P3_S4) { Reg x, y, z; }; // Register allocation of a P3_S4
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Component)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half(tx, base, offset + OA_(U2,[0])),
|
||||||
|
load_half(ty, base, offset + OA_(U2,[1])),
|
||||||
|
load_half(tz, base, offset + OA_(U2,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half(rs_x, r_base, offset + O_(V3_S2,x)),
|
||||||
|
load_half(rs_y, r_base, offset + O_(V3_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_word(tx, base, offset + OA_(U4,[0])),
|
||||||
|
load_word(ty, base, offset + OA_(U4,[1])),
|
||||||
|
load_word(tz, base, offset + OA_(U4,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
FI_ Slice_MipsCode ac_load_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_load_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_half_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(tx, base, offset + OA_(U2,[0])),
|
||||||
|
store_half(ty, base, offset + OA_(U2,[1])),
|
||||||
|
store_half(tz, base, offset + OA_(U2,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v3s2(AtomBuilder_R ab, Reg_(V3_S2) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_half_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_word_v3(AtomBuilder_R ab, Reg tx, Reg ty, Reg tz, Reg base, U2 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_word(tx, base, offset + OA_(U4,[0])),
|
||||||
|
store_word(ty, base, offset + OA_(U4,[1])),
|
||||||
|
store_word(tz, base, offset + OA_(U4,[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, Reg_(V3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
FI_ Slice_MipsCode ac_store_p3s4(AtomBuilder_R ab, Reg_(P3_S4) transfer, Reg base, U2 offset) MipsAtomComp_ProcMap_(ab, mac_store_word_v3(transfer.x, transfer.y, transfer.z, base, offset))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_add_si_v3s4(AtomBuilder_R ab, Reg rt_x, Reg rt_y, Reg rt_z, Reg base, U2 offset)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
add_si(rt_x, base, O_(V3_S4,x)),
|
||||||
|
add_si(rt_y, base, O_(V3_S4,y)),
|
||||||
|
add_si(rt_z, base, O_(V3_S4,z)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_s_v3(AtomBuilder_R ab
|
||||||
|
, Reg dx, Reg dy, Reg dz
|
||||||
|
, Reg sx, Reg sy, Reg sz
|
||||||
|
, Reg tx, Reg ty, Reg tz
|
||||||
|
) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
sub_s(dx, sx, tx),
|
||||||
|
sub_s(dy, sy, ty),
|
||||||
|
sub_s(dz, sz, tz),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, Reg_(V3_S4) d, Reg_(V3_S4) s, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3(d.x, d.y, d.z, s.x, s.y, s.z, t.x, t.y, t.z))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_s_v3_self(AtomBuilder_R ab, Reg ds_x, Reg ds_y, Reg ds_z, Reg tx, Reg ty, Reg tz) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
sub_s(ds_x, ds_x, tx),
|
||||||
|
sub_s(ds_y, ds_y, ty),
|
||||||
|
sub_s(ds_z, ds_z, tz),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg_(V3_S4) t) MipsAtomComp_ProcMap_(ab, mac_sub_s_v3_self(ds.x, ds.y, ds.z, t.x, t.y, t.z))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||||
|
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||||
|
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Component)
|
||||||
@@ -131,3 +131,15 @@ FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
|||||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||||
|
|
||||||
#pragma endregion FArena
|
#pragma endregion FArena
|
||||||
|
|
||||||
|
#pragma region BIOS Scratchpad
|
||||||
|
/* BIOS scratchpad location. 1 KB at 0x1F800000.
|
||||||
|
* TapeHostFrame occupies the final 44 bytes while tape code executes.
|
||||||
|
* Atom scratch is bounded by the TapeHostFrame_Loc declaration in lottes_tape.h. */
|
||||||
|
enum {
|
||||||
|
Scratchpad_Loc = 0x1F800000,
|
||||||
|
Scratchpad_Len = 0x400, /* 1 KB */
|
||||||
|
Scratchpad_End = Scratchpad_Loc + Scratchpad_Len, /* 0x1F800400 */
|
||||||
|
};
|
||||||
|
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
||||||
|
#pragma endregion BIOS Scratchpad
|
||||||
|
|||||||
@@ -16,6 +16,34 @@ atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
|||||||
or_i_self( dst, u4_lo(imm)),
|
or_i_self( dst, u4_lo(imm)),
|
||||||
})
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_v3_self(AtomBuilder_R ab, Reg dt_x, Reg dt_y, Reg dt_z, U2 shift_amount)
|
||||||
|
MipsAtomComp_Proc_( ab, {
|
||||||
|
shift_aright(dt_x, dt_x, shift_amount),
|
||||||
|
shift_aright(dt_y, dt_y, shift_amount),
|
||||||
|
shift_aright(dt_z, dt_z, shift_amount),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) dt, U2 shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_v3_self(dt.x, dt.y, dt.z, shift))
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
|
||||||
|
, Reg rd_v0, Reg rd_v1, Reg rd_v2
|
||||||
|
, Reg rs_v0, Reg rs_v1, Reg rs_v2
|
||||||
|
, Reg r_shift)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_aright_var(rd_v0, rs_v0, r_shift),
|
||||||
|
shift_aright_var(rd_v1, rs_v1, r_shift),
|
||||||
|
shift_aright_var(rd_v2, rs_v2, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab, Reg rds_v0, Reg rds_v1, Reg rds_v2, Reg r_shift)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_aright_var(rds_v0, rds_v0, r_shift),
|
||||||
|
shift_aright_var(rds_v1, rds_v1, r_shift),
|
||||||
|
shift_aright_var(rds_v2, rds_v2, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3s4_self(AtomBuilder_R ab, Reg_(V3_S4) ds, Reg shift) MipsAtomComp_ProcMap_(ab, mac_shift_aright_var_v3_self(ds.x, ds.y, ds.z, shift))
|
||||||
|
|
||||||
#pragma endregion MACs (Mips Atom Components)
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
#pragma region Baked Atoms
|
#pragma region Baked Atoms
|
||||||
|
|||||||
@@ -13,8 +13,7 @@ ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
|||||||
|
|
||||||
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
|
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
|
||||||
or_i_self( scratch, PadAxis_Centered & 0xFFFF),
|
or_i_self( scratch, PadAxis_Centered & 0xFFFF), // mac_load_word_imm(scratch, PadAxis_Centered),
|
||||||
// mac_load_word_imm(scratch, PadAxis_Centered),
|
|
||||||
store_word( scratch, state, O_(PadState,axes)),
|
store_word( scratch, state, O_(PadState,axes)),
|
||||||
})
|
})
|
||||||
|
|
||||||
|
|||||||
@@ -55,6 +55,7 @@ WORD_COUNT(gte_mv_to_ctrl_r, 1)
|
|||||||
WORD_COUNT(gte_sw, 1)
|
WORD_COUNT(gte_sw, 1)
|
||||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||||
|
WORD_COUNT(gte_cmdw_op, 1)
|
||||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||||
WORD_COUNT(gte_cmdw_sqr, 1)
|
WORD_COUNT(gte_cmdw_sqr, 1)
|
||||||
WORD_COUNT(gte_cmdw_gpf, 1)
|
WORD_COUNT(gte_cmdw_gpf, 1)
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
#pragma region hello_camera
|
#pragma region hello_camera
|
||||||
|
|
||||||
|
|
||||||
// --- atom: pad_input_cube_rotation (60 words) ---
|
// --- atom: pad_input_cube_rotation (61 words) ---
|
||||||
|
|
||||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||||
@@ -44,7 +44,7 @@ enum {
|
|||||||
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- atom: cube_g4_face (76 words) ---
|
// --- atom: cube_g4_face (75 words) ---
|
||||||
|
|
||||||
#define _atom_offset_cull_cube_g4_face_exit 41
|
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||||
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|||||||
@@ -10,7 +10,7 @@
|
|||||||
# include "duffle/pad.h"
|
# include "duffle/pad.h"
|
||||||
# include "duffle/word_count.metadata.h"
|
# include "duffle/word_count.metadata.h"
|
||||||
# include "duffle/psyq.h"
|
# include "duffle/psyq.h"
|
||||||
# include "duffle/math.atom.c"
|
# include "duffle/math.atom.h"
|
||||||
# include "duffle/mips.atom.c"
|
# include "duffle/mips.atom.c"
|
||||||
# include "duffle/gte.atom.c"
|
# include "duffle/gte.atom.c"
|
||||||
# include "duffle/gp.atom.c"
|
# include "duffle/gp.atom.c"
|
||||||
@@ -94,439 +94,187 @@ MipsAtomComp_Proc_(ab, {
|
|||||||
#pragma region Atom Procs
|
#pragma region Atom Procs
|
||||||
// Modular Atoms
|
// Modular Atoms
|
||||||
|
|
||||||
enum {
|
#define AtomBundle_(name) Struct_(tmpl(AtomBundle,name))
|
||||||
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
|
#define AtomBundle_Len(name) S_(tmpl(AtomBundle,name))/S_(MipsAtom*)
|
||||||
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
|
#define AtomBundleEntry_(bundle,entry) tmpl(bundle,entry)
|
||||||
#define R_ResolveScratch_Code R_T4_Code
|
|
||||||
|
#pragma region resolve_look_at
|
||||||
|
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
||||||
|
|
||||||
|
typedef AtomBundle_(resolve_look_at) { MipsAtom
|
||||||
|
*input_and_sub,
|
||||||
|
*normalize_fwd_uz,
|
||||||
|
*cross_to_right,
|
||||||
|
*normalize_right_ux,
|
||||||
|
*cross_to_up,
|
||||||
|
*normalize_up_uy,
|
||||||
|
*pop_mv_trans;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
typedef Struct_(ResolveLookAtScratch) {
|
||||||
|
V3_S4 fwd;
|
||||||
|
V3_S4 uz;
|
||||||
|
V3_S4 right;
|
||||||
|
V3_S4 ux;
|
||||||
|
V3_S4 up;
|
||||||
|
V3_S4 uy;
|
||||||
|
P3_S4 eye;
|
||||||
|
P3_S4 target;
|
||||||
|
V3_S4 up_in;
|
||||||
|
};
|
||||||
|
|
||||||
|
/* Binds_ResolveLookAtSub — what the C side pushes onto the tape before input_and_sub.
|
||||||
|
* The scratchpad base is no longer pushed because R_ScratchBase (= R_SP) is a tape carrier
|
||||||
|
* preserved across atoms; the atom body reads 0x1F800000 directly from R_SP. */
|
||||||
typedef Struct_(Binds_ResolveLookAt) {
|
typedef Struct_(Binds_ResolveLookAt) {
|
||||||
MT3_S2S4* look_at;
|
MT3_S2S4* look_at;
|
||||||
P3_S4* eye;
|
P3_S4* eye;
|
||||||
P3_S4* target;
|
P3_S4* target;
|
||||||
V3_S4* up_in;
|
V3_S4* up_in;
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's */
|
|
||||||
typedef Struct_(ResolveLookAtScratch) {
|
|
||||||
V3_S4 fwd; /* offset +0 (16 bytes — 4 S4 fields incl. internal pad) */
|
|
||||||
V3_S4 uz; /* offset +16 (16 bytes) */
|
|
||||||
V3_S4 right; /* offset +32 (16 bytes) */
|
|
||||||
V3_S4 ux; /* offset +48 (16 bytes) */
|
|
||||||
V3_S4 up; /* offset +64 (16 bytes) */
|
|
||||||
V3_S4 uy; /* offset +80 (16 bytes) */
|
|
||||||
P3_S4 eye; /* offset +96 (16 bytes; storage alias of V3_S4) */
|
|
||||||
P3_S4 target; /* offset +112 (16 bytes; storage alias of V3_S4) */
|
|
||||||
V3_S4 up_in; /* offset +128 (16 bytes) */
|
|
||||||
};
|
|
||||||
|
|
||||||
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
|
||||||
|
|
||||||
typedef Struct_(Binds_ResolveLookAtSub) {
|
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||||
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
|
P3_S4* target;
|
||||||
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
P3_S4* eye;
|
||||||
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
V3_S4* up_in;
|
||||||
ResolveLookAtScratch* scratchpad;
|
|
||||||
};
|
};
|
||||||
|
typedef Struct_(RegUse_resolve_look_at_input_and_sub) {
|
||||||
|
Reg target; Reg eye; Reg up_in;
|
||||||
|
Reg t0; Reg t1; Reg t2; Reg t3; Reg t4;
|
||||||
|
};
|
||||||
|
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye. */
|
||||||
|
internal MipsAtom* AtomBundleEntry_(resolve_look_at,input_and_sub)(AtomArena_R aa, RegUse_resolve_look_at_input_and_sub r)
|
||||||
|
atom_info(atom_bind(Binds_ResolveLookAtSub)) MipsAtom_Proc_(aa, {
|
||||||
|
load_word(r.target, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||||
|
load_word(r.eye, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||||
|
load_word(r.up_in, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||||
|
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||||
|
|
||||||
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
|
/* Stage up_in.x/y/z into the scratchpad. R_ScratchBase = R_SP = 0x1F800000. */
|
||||||
* Staging work:
|
mac_load_word_v3( r.t0, r.t1, r.t2, r.up_in, 0), LdSlot_
|
||||||
* * Stage eye.x/y/z → scratch (for atom 6's translation column)
|
mac_store_word_v3(r.t0, r.t1, r.t2, R_ScratchBase, O_(ResolveLookAtScratch,up_in)),
|
||||||
* * Stage up_in.x/y/z → scratch (for atom 2's outer-product operand)
|
|
||||||
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
|
|
||||||
* GPR codes (assigned by resolve_look_at_init):
|
|
||||||
* r_target_ptr : R_T0
|
|
||||||
* r_eye_ptr : R_T1
|
|
||||||
* r_up_in_ptr : R_T2
|
|
||||||
* r_scratch : R_T4 (R_ResolveScratch; wave-context carrier)
|
|
||||||
* r_tmp0 : R_T3 (stage eye/up_in + load eye.y)
|
|
||||||
* r_tmp1 : R_T5 (stage eye/up_in + load eye.z)
|
|
||||||
* r_tmp2 : R_T6 (stage eye/up_in + load target.x)
|
|
||||||
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
|
|
||||||
* R_AT : hardcoded (load eye.y / eye.z / target.z)
|
|
||||||
* R_V0 : hardcoded (load eye.z / target.z)
|
|
||||||
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
|
|
||||||
*/
|
|
||||||
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa,
|
|
||||||
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
|
|
||||||
U4 r_scratch
|
|
||||||
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
|
|
||||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
|
|
||||||
) MipsAtom_Proc_(aa, {
|
|
||||||
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
|
||||||
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
|
||||||
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
|
||||||
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
|
||||||
|
|
||||||
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
||||||
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_eye_ptr, 0),
|
mac_load_word_v3( r.t0, r.t1, r.t2, r.eye, 0), LdSlot_
|
||||||
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye)),
|
mac_store_word_v3(r.t0, r.t1, r.t2, R_ScratchBase, O_(ResolveLookAtScratch,eye)),
|
||||||
|
|
||||||
/* Stage up_in.x/y/z into the scratchpad. */
|
|
||||||
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_up_in_ptr, 0),
|
|
||||||
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in)),
|
|
||||||
|
|
||||||
/* Compute fwd = target - eye. */
|
/* Compute fwd = target - eye. */
|
||||||
mac_load_p3s4(r_tmp0, r_tmp1, r_tmp2, r_target_ptr, 0),
|
// mac_load_p3s4(t3, R_AT, t4, r.eye, 0),
|
||||||
mac_load_p3s4(r_tmp3, R_AT, R_V0, r_eye_ptr, 0),
|
mac_load_word_v3(r.t3, R_AT, r.t4, r.target, 0), LdSlot_
|
||||||
mac_sub_v3s4(
|
mac_sub_s_v3_self(
|
||||||
r_tmp0, r_tmp1, r_tmp2,
|
r.t3, R_AT, r.t4,
|
||||||
r_tmp3, R_AT, R_V0),
|
r.t0, r.t1, r.t2),
|
||||||
mac_store_v3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd)),
|
mac_store_word_v3(r.t3, R_AT, r.t4, R_ScratchBase, O_(ResolveLookAtScratch,fwd)),
|
||||||
|
|
||||||
mac_yield()
|
mac_yield()
|
||||||
})
|
})
|
||||||
|
|
||||||
/* Atom 2: cross uz × up_in → right. */
|
|
||||||
internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch
|
|
||||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
|
||||||
, U4 r_d /* load b.x */
|
|
||||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
|
|
||||||
) MipsAtom_Proc_(aa, {
|
|
||||||
/* FIX: build packed RT22+RT33 with proper sign extension. */
|
|
||||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
|
||||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
|
|
||||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
typedef Struct_(Binds_ResolveLookAtPopMvTrans) {
|
||||||
load_word(r_a, r_g, O_(V3_S4,x)),
|
U4 look_at; /* MT3_S2S4* — destination matrix address */
|
||||||
load_word(r_b, r_g, O_(V3_S4,y)),
|
|
||||||
load_word(r_c, r_g, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
|
|
||||||
load_word(r_d, r_h, O_(V3_S4,x)),
|
|
||||||
load_word(R_AT, r_h, O_(V3_S4,y)),
|
|
||||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* Save the two RT control-register slots OP will clobber. We reuse
|
|
||||||
* r_g/r_h (scratch pointers, no longer needed) as the save targets. */
|
|
||||||
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
|
|
||||||
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
|
|
||||||
|
|
||||||
/* Load uz.x/uz.y/uz.z into COP2 control registers.
|
|
||||||
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
|
||||||
* RT22 is in BOTH $2.high AND $4.low (shared bit position). OP reads from $2.high.
|
|
||||||
* So set RT22 via ctc2 r_b, $2 (sets $2.high = a.y.high = RT22, $2.low = a.y.low = RT13).
|
|
||||||
* Then set RT33 via ctc2 r_c, $4 (sets $4.high = a.z.high = RT33, $4.low = a.z.low).
|
|
||||||
* The $2 and $4 writes don't clobber each other (separate registers).
|
|
||||||
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
|
|
||||||
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
|
|
||||||
* This is libpsyx's OuterProduct12 convention EXACTLY. */
|
|
||||||
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
|
|
||||||
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
|
|
||||||
|
|
||||||
/* Load uz into the RT diagonal. */
|
|
||||||
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
|
|
||||||
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
|
||||||
|
|
||||||
/* Load up_in into IR (the second operand for OP). */
|
|
||||||
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = up_in.x */
|
|
||||||
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
|
|
||||||
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = up_in.z */
|
|
||||||
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
|
||||||
|
|
||||||
gte_cmdw_outer_product, /* OP: MAC1/2/3 = uz × up_in
|
|
||||||
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
|
|
||||||
* MAC2 = IR1*D3 - IR3*D1 = up_in.x*uz.z.high - up_in.z*uz.x
|
|
||||||
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
|
|
||||||
* For up_in = (0, -fp_one, 0):
|
|
||||||
* MAC1 = 0 - (-fp_one)*uz.z.high = fp_one*uz.z.high
|
|
||||||
* MAC2 = 0 - 0 = 0
|
|
||||||
* MAC3 = (-fp_one)*uz.x - 0 = -fp_one*uz.x */
|
|
||||||
|
|
||||||
/* Restore the RT slots we clobbered. */
|
|
||||||
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
|
|
||||||
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
|
|
||||||
|
|
||||||
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
|
|
||||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(r_b, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
|
||||||
nop, /* MFC2 retirement */
|
|
||||||
|
|
||||||
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
|
|
||||||
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
|
|
||||||
shift_aright(r_a, r_a, 12),
|
|
||||||
shift_aright(r_b, r_b, 12),
|
|
||||||
shift_aright(r_c, r_c, 12),
|
|
||||||
|
|
||||||
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
|
|
||||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
|
||||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
|
||||||
store_word(r_c, r_f, O_(V3_S4,z)),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Atom 4: cross uz × ux → up. */
|
|
||||||
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
|
|
||||||
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
|
||||||
, U4 r_d /* load b.x */
|
|
||||||
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
|
|
||||||
) MipsAtom_Proc_(aa, {
|
|
||||||
/* Compute the three scratch pointers from r_scratch. */
|
|
||||||
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
|
||||||
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
|
|
||||||
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
|
||||||
load_word(r_a, r_g, O_(V3_S4,x)),
|
|
||||||
load_word(r_b, r_g, O_(V3_S4,y)),
|
|
||||||
load_word(r_c, r_g, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
|
|
||||||
load_word(r_d, r_h, O_(V3_S4,x)),
|
|
||||||
load_word(R_AT, r_h, O_(V3_S4,y)),
|
|
||||||
load_word(R_V0, r_h, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
|
|
||||||
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
|
|
||||||
* ctc2 RT restore. */
|
|
||||||
|
|
||||||
/* Save the two RT control-register slots OP will clobber (reusing
|
|
||||||
* r_g/r_h — they're no longer needed as scratch pointers). */
|
|
||||||
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
|
|
||||||
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
|
|
||||||
|
|
||||||
/* Load uz into the RT diagonal — same packing as atom 1.
|
|
||||||
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
|
||||||
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
|
|
||||||
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
|
|
||||||
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
|
|
||||||
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
|
|
||||||
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
|
|
||||||
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
|
|
||||||
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
|
|
||||||
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
|
||||||
|
|
||||||
/* Load ux into the IR registers (the second operand for OP). */
|
|
||||||
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
|
|
||||||
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
|
|
||||||
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
|
|
||||||
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
|
||||||
|
|
||||||
gte_cmdw_outer_product,
|
|
||||||
|
|
||||||
/* Restore the RT slots we clobbered. */
|
|
||||||
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
|
|
||||||
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
|
|
||||||
|
|
||||||
gte_mv_from_data_r(r_a, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(r_b, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(r_c, C2_MAC3),
|
|
||||||
nop,
|
|
||||||
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
|
|
||||||
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
|
|
||||||
shift_aright(r_a, r_a, 12),
|
|
||||||
shift_aright(r_b, r_b, 12),
|
|
||||||
shift_aright(r_c, r_c, 12),
|
|
||||||
store_word(r_a, r_f, O_(V3_S4,x)),
|
|
||||||
store_word(r_b, r_f, O_(V3_S4,y)),
|
|
||||||
store_word(r_c, r_f, O_(V3_S4,z)),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
|
|
||||||
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
|
|
||||||
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
|
|
||||||
};
|
};
|
||||||
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye).
|
typedef Struct_(RegUse_resolve_look_at__pop_mv_trans) {
|
||||||
|
Reg look_at;
|
||||||
|
Reg_(V3_S4) row; /* populate phase: load ux/uy/uz */
|
||||||
|
union { Reg ux, v_x; } t6; /* populate addr (canonical) → matrix_vector v_x */
|
||||||
|
union { Reg uy, v_y; } t7; /* populate uy → matrix_vector v_y */
|
||||||
|
union { Reg uz, v_z; } t8; /* populate uz → matrix_vector v_z */
|
||||||
|
Reg eye; /* matrix_vector phase: load -eye */
|
||||||
|
};
|
||||||
|
/* Atom 6 (fused): write look_at->m[][] from ux/uy/uz as packed S2 (populate),
|
||||||
|
* ctc2 RT chain into C2[0..4] (matrix_vector), MVMVA RT*(-eye)>>12, store off
|
||||||
|
* directly to look_at->t[] (trans_matrix). Replaces the previous 3 separate atoms
|
||||||
|
* (populate + matrix_vector + trans_matrix).
|
||||||
*
|
*
|
||||||
* GPR codes (assigned by resolve_look_at_init):
|
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; }
|
||||||
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
* m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
||||||
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
|
* t[0..2] is S4 (3 × 4 = 12 bytes at offset 18)
|
||||||
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
|
|
||||||
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
|
|
||||||
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
|
|
||||||
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
|
|
||||||
*
|
*
|
||||||
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
|
* C11 ApplyMatrixLV semantics (gte.atom.c ac_apply_matrix_lv; libgte reference):
|
||||||
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
|
|
||||||
*
|
|
||||||
* Struct layout (per duffle/math.h):
|
|
||||||
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
|
||||||
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
|
|
||||||
*
|
|
||||||
* Translation column: GTE MVMVA with the world rotation matrix pre-set
|
|
||||||
* (helper emits set_gte_world before the bundle, per the bundle design).
|
|
||||||
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
|
|
||||||
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
|
|
||||||
*/
|
|
||||||
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
|
|
||||||
, U4 r_look_at
|
|
||||||
, U4 r_scratch
|
|
||||||
, U4 r_pux, U4 r_puy, U4 r_puz
|
|
||||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
|
||||||
) MipsAtom_Proc_(aa, {
|
|
||||||
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
|
|
||||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
|
||||||
|
|
||||||
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
|
|
||||||
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
|
|
||||||
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
|
|
||||||
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* ── m[0] = (S2)ux ── */
|
|
||||||
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
|
|
||||||
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
|
|
||||||
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
|
|
||||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
|
|
||||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
|
|
||||||
|
|
||||||
/* ── m[1] = (S2)uy ── */
|
|
||||||
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
|
|
||||||
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
|
|
||||||
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
|
|
||||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
|
|
||||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
|
|
||||||
|
|
||||||
/* ── m[2] = (S2)uz ── */
|
|
||||||
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
|
|
||||||
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
|
|
||||||
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
|
|
||||||
nop,
|
|
||||||
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
|
|
||||||
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
|
|
||||||
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
|
|
||||||
|
|
||||||
/* Zero t[0..2] — atom 6c writes the final values here. */
|
|
||||||
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
|
|
||||||
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
|
|
||||||
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
|
|
||||||
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
|
|
||||||
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
|
|
||||||
* Stores off to scratch+96 (overwriting the packed pos).
|
|
||||||
*
|
|
||||||
* GPR codes (assigned by resolve_look_at_init):
|
|
||||||
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
|
||||||
* r_peye : pointer to eye (slot +96, reused as off destination)
|
|
||||||
* r_tmp0/1/2: -eye + GTE transfer scratch
|
|
||||||
*
|
|
||||||
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
|
|
||||||
*/
|
|
||||||
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
|
|
||||||
, U4 r_scratch
|
|
||||||
, U4 r_peye
|
|
||||||
, U4 r_look_at
|
|
||||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
|
||||||
) MipsAtom_Proc_(aa, {
|
|
||||||
/* === EXACT C11 ApplyMatrixLV replication ===
|
|
||||||
* The C11 does:
|
|
||||||
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
||||||
* 2. lw v.x/y/z from memory
|
* 2. lw -eye from memory
|
||||||
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
|
* 3. S15 decomposition (eliminated here — the fused body takes the >>12 path
|
||||||
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3)
|
* directly via mtc2 IR + MVMVA pass2, matching the libgte canonical output)
|
||||||
* 5. mfc2 MACs
|
* 4. mtc2 to IR1/2/3, nop2, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
||||||
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
* 5. mfc2 MACs → off
|
||||||
* 7. mfc2 MACs
|
* 6. store off to look_at->t[] (skip scratch.eye intermediate)
|
||||||
* 8. Combine: (pass1 << 3) + pass2
|
|
||||||
*
|
|
||||||
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
|
|
||||||
* The combine simplifies: result = 0 + pass2 = pass2.
|
|
||||||
* So we skip the S15 decomposition and just do pass 2 directly.
|
|
||||||
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
|
|
||||||
|
|
||||||
/* Pop look_at* from tape. */
|
|
||||||
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
|
||||||
|
|
||||||
/* r_peye = &eye (slot +96, reused as off destination). */
|
|
||||||
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
|
|
||||||
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
|
|
||||||
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
|
|
||||||
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
|
|
||||||
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
|
|
||||||
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
|
|
||||||
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
|
|
||||||
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
|
|
||||||
|
|
||||||
/* Load pos = -eye after the matrix load releases r_tmp0. */
|
|
||||||
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
|
|
||||||
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
|
|
||||||
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
|
|
||||||
nop,
|
|
||||||
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
|
|
||||||
sub_u(r_tmp1, R_0, r_tmp1),
|
|
||||||
sub_u(r_tmp2, R_0, r_tmp2),
|
|
||||||
|
|
||||||
/* === mtc2 pos (as S16) to IR1/2/3 ===
|
|
||||||
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
|
|
||||||
* 32-bit sign-extended value's low 16 bits = correct S16. */
|
|
||||||
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
|
|
||||||
* gives the correct S16 value (sign bit preserved). */
|
|
||||||
/* r_tmp0/1/2 already have pos values. */
|
|
||||||
gte_mv_to_data_r(r_tmp0, C2_IR1),
|
|
||||||
gte_mv_to_data_r(r_tmp1, C2_IR2),
|
|
||||||
gte_mv_to_data_r(r_tmp2, C2_IR3),
|
|
||||||
nop2, /* MTC2 retirement (2 slots) */
|
|
||||||
|
|
||||||
/* === MVMVA pass 2 — C11 ApplyMatrixLV command ===
|
|
||||||
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
|
||||||
gte_cmdw_mvmva_c11_pass2,
|
|
||||||
nop, /* GTE interlock */
|
|
||||||
|
|
||||||
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
|
|
||||||
gte_mv_from_data_r(r_tmp0, C2_MAC1),
|
|
||||||
gte_mv_from_data_r(r_tmp1, C2_MAC2),
|
|
||||||
gte_mv_from_data_r(r_tmp2, C2_MAC3),
|
|
||||||
nop,
|
|
||||||
|
|
||||||
/* === Store off → scratch+96 (overwriting pos) === */
|
|
||||||
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
|
|
||||||
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
|
|
||||||
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
|
|
||||||
|
|
||||||
mac_yield()
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Atom 6c in the bundle: copy scratch+96 (off, written by atom 6b) → look_at->t[].
|
|
||||||
* Uses mac_trans_matrix component (m->t = v, libgte TransMatrix semantics = struct copy).
|
|
||||||
*
|
*
|
||||||
* GPR codes (assigned by resolve_look_at_init):
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
* r_scratch : R_ResolveScratch (R_T4 carrier)
|
||||||
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
* r_look_at : ralloc() — also serves as the off-dst in the trans_matrix phase
|
||||||
* r_off_ptr : pointer to off (= &scratch.eye, reused slot)
|
* r_row : V3_S4, reused for ux/uy/uz loads in populate phase
|
||||||
* r_tmp0 : transfer reg for mac_trans_matrix
|
* r_eye : ralloc() — &scratch.eye, used for -eye load in matrix_vector phase
|
||||||
|
* r_v_x/v_y/v_z : ralloc() — populate scratch addrs (ux/uy/uz), reused as
|
||||||
|
* ctc2 transfer + MVMVA -eye temp in matrix_vector phase
|
||||||
|
* (v_x/v_y/v_z alias ux/uy/uz via the union; lifetime ends for ux/uy/uz after
|
||||||
|
* populate's mac_load_v3s4, so reusing for v.x/v.y/v.z is safe)
|
||||||
|
* Pool cost: 1 carrier + 1 look_at + 3 row + 1 eye + 3 aliased = 9 GPRs
|
||||||
*
|
*
|
||||||
* Pool cost: r_look_at (1) + r_scratch (carrier) + r_off_ptr + 1 clobber = 4 GPRs.
|
* Net word savings vs the previous 3-atom flow: ~15 words + 2 mac_yields + 1 tape pop.
|
||||||
|
* - 2 mac_yields (trans_matrix's + matrix_vector's) → fused into one yield
|
||||||
|
* - 1 redundant tb_data (look_at was pushed 2x; now once)
|
||||||
|
* - mac_trans_mt3s3s4 (6 words) → replaced by direct mac_store_v3s4
|
||||||
|
* - mac_store_v3s4 to scratch.eye (3 words intermediate) → eliminated
|
||||||
|
* - add_si for r_off_ptr (2 words) → eliminated
|
||||||
|
* - mac_store_v3s4 zero-store of t[] (3 words) → eliminated (matrix_vector writes
|
||||||
|
* off directly; no consumer needed the zero first)
|
||||||
|
* - 1 set_gte_mt3s2s4 ctc2 chain (13 baked words) → eliminated (matrix_vector
|
||||||
|
* has its own ctc2 RT chain; cube rendering atoms reload C2 state themselves)
|
||||||
*/
|
*/
|
||||||
I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa
|
internal MipsAtom* resolve_look_at__pop_mv_trans(AtomArena_R aa,
|
||||||
, U4 r_look_at, U4 r_scratch, U4 r_off_ptr
|
RegUse_resolve_look_at__pop_mv_trans r
|
||||||
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
|
||||||
) MipsAtom_Proc_(aa, {
|
) MipsAtom_Proc_(aa, {
|
||||||
/* Pop look_at* from tape. */
|
/* --- Tape pop: look_at pointer --- */
|
||||||
// load_word(r_Vlook_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
load_word(r.look_at, R_TapePtr, O_(Binds_ResolveLookAtPopMvTrans,look_at)),
|
||||||
// add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
LdSlot_ add_ui_self(R_TapePtr, S_(Binds_ResolveLookAtPopMvTrans)),
|
||||||
|
|
||||||
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */
|
/* --- Scratch addresses for ux/uy/uz/eye (populate phase; t6/t7/t8 alias ux/uy/uz).
|
||||||
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)),
|
* R_ScratchBase (= R_SP) holds 0x1F800000; no per-atom bake is required because
|
||||||
nop,
|
* R_SP is a tape carrier preserved across atoms. --- */
|
||||||
|
add_si(r.t6.ux, R_ScratchBase, O_(ResolveLookAtScratch, ux)), LdSlot_
|
||||||
|
add_si(r.t7.uy, R_ScratchBase, O_(ResolveLookAtScratch, uy)),
|
||||||
|
add_si(r.t8.uz, R_ScratchBase, O_(ResolveLookAtScratch, uz)),
|
||||||
|
add_si(r.eye, R_ScratchBase, O_(ResolveLookAtScratch, eye)),
|
||||||
|
|
||||||
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */
|
/* --- POPULATE phase: write look_at->m[][] from ux/uy/uz as packed S2 --- */
|
||||||
mac_trans_mt3s3s4(r_look_at, r_off_ptr, r_tmp0, r_tmp1, r_tmp2),
|
mac_load_v3s4(r.row, r.t6.ux, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[0])),
|
||||||
|
mac_load_v3s4(r.row, r.t7.uy, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[1])),
|
||||||
|
mac_load_v3s4(r.row, r.t8.uz, 0), LdSlot_ mac_store_v3s2(r.row, r.look_at, O_(MT3_S2S4, m[2])),
|
||||||
|
|
||||||
|
/* --- MATRIX-VECTOR phase: ctc2 RT chain + MVMVA RT*(-eye)>>12 --- */
|
||||||
|
/* RT packing (per libgte ApplyMatrixLV convention; see gte.h:217-220 +
|
||||||
|
* atom_6b_disasm_comparison.md:28-32):
|
||||||
|
* C2[0] = (RT12<<16)|RT11 ← ctc2 RT11 from m[0][0..1] packed word
|
||||||
|
* C2[1] = (RT21<<16)|RT13 ← ctc2 RT12 from m[0][2..3] packed word
|
||||||
|
* C2[2] = (RT23<<16)|RT22 ← ctc2 RT13 from m[1][1..2] packed word
|
||||||
|
* C2[3] = (RT32<<16)|RT31 ← ctc2 RT21 from m[2][0..1] packed word
|
||||||
|
* C2[4] = (RT33<<16)|junk ← ctc2 RT22 from m[2][2] (half)
|
||||||
|
* Each ctc2 writes a WHOLE 32-bit C2 slot; the "macro name" identifies
|
||||||
|
* which C2 register, not which 16-bit half. */
|
||||||
|
load_word( r.t6.v_x, r.look_at, O_(MT3_S2S4, m[0][0])), /* RT11|RT12 */ LdSlot_
|
||||||
|
load_word( r.t7.v_y, r.look_at, O_(MT3_S2S4, m[0][2])), /* RT13|RT21 */ LdSlot_ gte_mv_to_ctrl_r(r.t6.v_x, gte_cr_RT11),
|
||||||
|
load_word( r.t8.v_z, r.look_at, O_(MT3_S2S4, m[1][1])), /* RT22|RT23 */ LdSlot_ gte_mv_to_ctrl_r(r.t7.v_y, gte_cr_RT12),
|
||||||
|
load_word( r.t6.v_x, r.look_at, O_(MT3_S2S4, m[2][0])), /* RT31|RT32 */ LdSlot_ gte_mv_to_ctrl_r(r.t8.v_z, gte_cr_RT13),
|
||||||
|
load_half_u(r.t7.v_y, r.look_at, O_(MT3_S2S4, m[2][2])), /* RT33 */ LdSlot_ gte_mv_to_ctrl_r(r.t6.v_x, gte_cr_RT21),
|
||||||
|
/* pos = -eye. The three loads also retire the last CTC2. */ gte_mv_to_ctrl_r(r.t7.v_y, gte_cr_RT22),
|
||||||
|
GteDelay_ mac_load_word_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, r.eye, 0), LdSlot_
|
||||||
|
mac_sub_s_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, R_0, R_0, R_0, r.t6.v_x, r.t7.v_y, r.t8.v_z),
|
||||||
|
/* mtc2 pos (as S16) to IR1/2/3. The GTE takes low 16 bits. pos fits in S16. For negative pos, the 32-bit sign-extended value's low 16 bits = correct S16. */
|
||||||
|
gte_mv_to_data_r(r.t6.v_x, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r.t7.v_y, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r.t8.v_z, C2_IR3),
|
||||||
|
GteDelay_ nop2,
|
||||||
|
|
||||||
|
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
||||||
|
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
||||||
|
gte_cmdw_mvmva_c11_pass2, GteDelay_ nop,
|
||||||
|
mac_gte_mv_from_data_r_mac123(r.t6.v_x, r.t7.v_y, r.t8.v_z), GteDelay_ nop,
|
||||||
|
|
||||||
|
/* --- TRANS-MATRIX phase: store off directly to look_at->t[] (skip scratch.eye intermediate) --- */
|
||||||
|
mac_store_word_v3(r.t6.v_x, r.t7.v_y, r.t8.v_z, r.look_at, O_(MT3_S2S4, t)),
|
||||||
|
|
||||||
mac_yield()
|
mac_yield()
|
||||||
})
|
})
|
||||||
|
#pragma endregion resolve_look_at
|
||||||
|
|
||||||
#pragma endregion Atom Procs
|
#pragma endregion Atom Procs
|
||||||
|
|
||||||
@@ -658,15 +406,15 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||||
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||||
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||||
|
|
||||||
/* Load pad[0].buttons into R_T0. */
|
/* Load pad[0].buttons into R_T0. */
|
||||||
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), LdSlot_ nop,
|
||||||
// Note(Ed): Potential op with delay slot?
|
// Note(Ed): Potential op with delay slot?
|
||||||
|
|
||||||
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)), BdSlot_
|
||||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
add_si( R_T4, R_T4, 30),
|
add_si( R_T4, R_T4, 30),
|
||||||
add_si( R_T3, R_T3, 5),
|
add_si( R_T3, R_T3, 5),
|
||||||
@@ -675,8 +423,8 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
atom_label(exit_dpad_left)
|
atom_label(exit_dpad_left)
|
||||||
|
|
||||||
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)), BdSlot_
|
||||||
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), LdSlot_
|
||||||
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
add_si( R_T4, R_T4, -30),
|
add_si( R_T4, R_T4, -30),
|
||||||
add_si( R_T3, R_T3, -5),
|
add_si( R_T3, R_T3, -5),
|
||||||
@@ -686,7 +434,7 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
|
|
||||||
/* Analog left-stick X: dead zone 0x70..0x90.
|
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||||
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), LdSlot_ //?
|
||||||
|
|
||||||
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||||
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||||
@@ -695,14 +443,14 @@ internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyIn
|
|||||||
|
|
||||||
atom_label(dead_check_upper)
|
atom_label(dead_check_upper)
|
||||||
/* left_x >= 0x70 → check upper bound. */
|
/* left_x >= 0x70 → check upper bound. */
|
||||||
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */ LdSlot_ //?
|
||||||
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
||||||
|
|
||||||
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||||
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)), BdSlot_
|
||||||
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||||
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||||
mac_yield_load(),
|
BdSlot_ mac_yield_load(), LdSlot_
|
||||||
|
|
||||||
atom_label(dead_low_active)
|
atom_label(dead_low_active)
|
||||||
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
@@ -713,18 +461,18 @@ atom_label(dead_low_active)
|
|||||||
|
|
||||||
/* R_T4 = cube_delta */
|
/* R_T4 = cube_delta */
|
||||||
shift_aright(R_T4, R_T3, 2),
|
shift_aright(R_T4, R_T3, 2),
|
||||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||||
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||||
shift_aright(R_T4, R_T3, 5),
|
shift_aright(R_T4, R_T3, 5),
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
jump_rel(atom_offset(end_low, exit_stick)),
|
jump_rel(atom_offset(end_low, exit_stick)),
|
||||||
mac_yield_load(),
|
BdSlot_ mac_yield_load(), LdSlot_
|
||||||
|
|
||||||
atom_label(dead_high_active)
|
atom_label(dead_high_active)
|
||||||
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
@@ -734,18 +482,18 @@ atom_label(dead_high_active)
|
|||||||
/* delta = 0x80 - left_x (signed negative). */
|
/* delta = 0x80 - left_x (signed negative). */
|
||||||
|
|
||||||
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||||
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), LdSlot_ nop,
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||||
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)), LdSlot_
|
||||||
shift_aright(R_T4, R_T3, 5),
|
shift_aright(R_T4, R_T3, 5),
|
||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
atom_label(no_jump_fallthrough)
|
atom_label(no_jump_fallthrough)
|
||||||
mac_yield_load(),
|
mac_yield_load(), LdSlot_
|
||||||
|
|
||||||
atom_label(exit_stick)
|
atom_label(exit_stick)
|
||||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||||
@@ -767,41 +515,41 @@ internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
|||||||
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||||
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
||||||
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||||
|
|
||||||
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
||||||
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
|
load_word(R_T0, R_CamPadState, O_(PadState,buttons)), LdSlot_
|
||||||
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
|
load_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
|
|
||||||
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
||||||
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
|
LdSlot_ and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), BdSlot_ mac_yield_load(), LdSlot_
|
||||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
atom_label(exit_left_x)
|
atom_label(exit_left_x)
|
||||||
|
|
||||||
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
||||||
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
|
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), BdSlot_ nop,
|
||||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
atom_label(exit_right_x)
|
atom_label(exit_right_x)
|
||||||
|
|
||||||
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
||||||
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
load_word(R_T1, R_Cam, O_(Camera,pos.y)), LdSlot_
|
||||||
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
|
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), BdSlot_ nop,
|
||||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
atom_label(exit_up_y)
|
atom_label(exit_up_y)
|
||||||
|
|
||||||
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
||||||
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
|
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), BdSlot_ nop,
|
||||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
atom_label(exit_down_y)
|
atom_label(exit_down_y)
|
||||||
|
|
||||||
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
||||||
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
load_word(R_T1, R_Cam, O_(Camera,pos.z)), LdSlot_
|
||||||
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
|
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), BdSlot_ nop,
|
||||||
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
atom_label(exit_cross_z)
|
atom_label(exit_cross_z)
|
||||||
|
|
||||||
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
||||||
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
|
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), BdSlot_ nop,
|
||||||
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
atom_label(exit_circle_z)
|
atom_label(exit_circle_z)
|
||||||
|
|
||||||
@@ -833,7 +581,7 @@ internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_
|
|||||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||||
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||||
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||||
mac_yield()
|
mac_yield()
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -846,20 +594,20 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
|||||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
// load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
LdSlot_ mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2), GteDelay_ load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)), LdSlot_
|
||||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
GteDelay_ nop, gte_cmdw_rotate_translate_perspective_triple,
|
||||||
gte_cmdw_nclip,
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
gte_mv_from_data_r(R_T0, C2_MAC0), GteDelay_ nop,
|
||||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||||
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||||
* harmless because the OT entry that points to this prim is created later. */
|
* harmless because the OT entry that points to this prim is created later. */
|
||||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
BdSlot_ store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)), LdSlot_
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
mac_gte_store_g4_p012(R_PrimCursor),
|
mac_gte_store_g4_p012(R_PrimCursor),
|
||||||
@@ -871,7 +619,7 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
|||||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
set_lt_u( R_AT, R_T1, R_AT),
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
|
||||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), BdSlot_ nop,
|
||||||
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||||
mac_format_g4_color(R_PrimCursor,
|
mac_format_g4_color(R_PrimCursor,
|
||||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
@@ -903,7 +651,7 @@ MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(f
|
|||||||
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||||
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||||
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||||
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
LdSlot_ add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||||
mac_yield()
|
mac_yield()
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -950,7 +698,7 @@ internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitive
|
|||||||
, atom_writes(R_TapePtr)
|
, atom_writes(R_TapePtr)
|
||||||
){
|
){
|
||||||
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||||
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)), LdSlot_
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||||
/* Calculate byte offset and store directly back to RAM */
|
/* Calculate byte offset and store directly back to RAM */
|
||||||
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||||
|
|||||||
+110
-196
@@ -32,7 +32,7 @@
|
|||||||
|
|
||||||
#pragma region Duffle TUs
|
#pragma region Duffle TUs
|
||||||
#include "duffle/pad.c"
|
#include "duffle/pad.c"
|
||||||
#include "duffle/math.atom.c"
|
#include "duffle/math.atom.h"
|
||||||
#include "duffle/mips.atom.c"
|
#include "duffle/mips.atom.c"
|
||||||
#include "duffle/gte.atom.c"
|
#include "duffle/gte.atom.c"
|
||||||
#include "duffle/gp.atom.c"
|
#include "duffle/gp.atom.c"
|
||||||
@@ -53,15 +53,13 @@
|
|||||||
#pragma endregion Hello Joypad TUs
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
Scratchpad_Loc = 0x1F800000,
|
|
||||||
};
|
|
||||||
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
|
||||||
|
|
||||||
enum {
|
|
||||||
Scratchpad_Len = 1024,
|
|
||||||
MemTape_Len = 512,
|
MemTape_Len = 512,
|
||||||
|
|
||||||
ResolveLookAtArena_Words = 1024,
|
ResolveLookAtArena_Words = 1024,
|
||||||
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
||||||
|
|
||||||
|
CT_InitAtomMem_Words = Kilo_(4),
|
||||||
|
CT_InitAtomMem_Size = CT_InitAtomMem_Words * S_(MipsCode),
|
||||||
};
|
};
|
||||||
typedef Struct_(SMemory) {
|
typedef Struct_(SMemory) {
|
||||||
PrimitiveArena primitives;
|
PrimitiveArena primitives;
|
||||||
@@ -85,8 +83,12 @@ typedef Struct_(SMemory) {
|
|||||||
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
||||||
U4_V scratchpad; // d-cache
|
U4_V scratchpad; // d-cache
|
||||||
|
|
||||||
|
U1 ct_init_atom_mem[CT_InitAtomMem_Size];
|
||||||
|
MipsAtom* normalize_v3s4;
|
||||||
|
MipsAtom* gte_cross_v3s4;
|
||||||
|
|
||||||
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
||||||
MipsAtom* resolve_look_at_atom_addrs[10];
|
MipsAtom* resolve_look_at_bundle[AtomBundle_Len(resolve_look_at)];
|
||||||
};
|
};
|
||||||
global SMemory smem;
|
global SMemory smem;
|
||||||
extern SMemory smem;
|
extern SMemory smem;
|
||||||
@@ -131,203 +133,125 @@ I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4*
|
|||||||
}
|
}
|
||||||
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||||
|
|
||||||
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
|
internal void compile_init_atoms(void) {
|
||||||
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
|
AtomArena ab = atomarena_make(slice_ut_arr(smem.ct_init_atom_mem));
|
||||||
* share the GENERIC normalize_v3s4_proc from gte.atom.c
|
RegFile rf = regfile(regfile_abi_mask);
|
||||||
* 0: resolve_look_at__input_and_sub_proc
|
#define ralloc() regfile_alloc(& rf)
|
||||||
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
|
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||||
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
|
|
||||||
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
|
smem.gte_cross_v3s4 = gte_cross_v3s4(& ab,
|
||||||
* 4: resolve_look_at__cross_uz_ux_to_up_proc
|
RegUse_(gte_cross_v3s4) {
|
||||||
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
|
.a = ralloc_v3(),
|
||||||
* 6: resolve_look_at__populate_and_translate_proc
|
.b = ralloc_v3(),
|
||||||
*/
|
.x = ralloc(),
|
||||||
internal void resolve_look_at_init(void) {
|
.y = ralloc(),
|
||||||
/* Wrap the static arena in a MipsAtomBuilder. */
|
.z = ralloc(),
|
||||||
|
});
|
||||||
|
regfile_reset(& rf);
|
||||||
|
|
||||||
|
smem.normalize_v3s4 = build_normalize_v3s4(& ab,
|
||||||
|
RegUse_(build_normalize_v3s4) {
|
||||||
|
.scratch = ralloc(),
|
||||||
|
.src_ptr = ralloc(),
|
||||||
|
.dst_ptr = ralloc(),
|
||||||
|
.recip_est = ralloc(),
|
||||||
|
.norm = ralloc(),
|
||||||
|
.shift = ralloc(),
|
||||||
|
.src_x = ralloc(),
|
||||||
|
// .shift_count = ralloc(), /* dedicated slot for stage-3 → stage-4 shift count */
|
||||||
|
.t3 = ralloc(),
|
||||||
|
.t4 = ralloc(),
|
||||||
|
.t5 = ralloc(),
|
||||||
|
});
|
||||||
|
regfile_reset(& rf);
|
||||||
|
|
||||||
|
assert(ab.used <= CT_InitAtomMem_Size);
|
||||||
|
#undef ralloc
|
||||||
|
#undef ralloc_v3
|
||||||
|
}
|
||||||
|
|
||||||
|
internal void compile_resolve_look_at(void) {
|
||||||
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||||
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_atom_addrs));
|
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||||
|
|
||||||
U4 pin_mask = regfile_abi_mask | (1 << R_ResolveScratch);
|
/* R_ScratchBase (= R_SP) is a tape carrier preserved across atoms; no carrier
|
||||||
RegFile rf = regfile(pin_mask);
|
* pin is needed in the regfile. The standard 24-register pool is sufficient. */
|
||||||
|
RegFile rf = regfile(regfile_abi_mask);
|
||||||
|
#define ralloc() regfile_alloc(& rf)
|
||||||
|
#define ralloc_v3() { ralloc(), ralloc(), ralloc() }
|
||||||
|
|
||||||
U4 r_target_ptr = regfile_alloc(& rf);
|
bundle->input_and_sub = AtomBundleEntry_(resolve_look_at, input_and_sub)(& ab,
|
||||||
U4 r_eye_ptr = regfile_alloc(& rf);
|
RegUse_(resolve_look_at_input_and_sub) {
|
||||||
U4 r_up_in_ptr = regfile_alloc(& rf);
|
.target = ralloc(),
|
||||||
U4 r_tmp0 = regfile_alloc(& rf);
|
.eye = ralloc(),
|
||||||
U4 r_tmp1 = regfile_alloc(& rf);
|
.up_in = ralloc(),
|
||||||
U4 r_tmp2 = regfile_alloc(& rf);
|
.t0 = ralloc(),
|
||||||
U4 r_tmp3 = regfile_alloc(& rf);
|
.t1 = ralloc(),
|
||||||
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
|
.t2 = ralloc(),
|
||||||
R_ResolveScratch,
|
.t3 = ralloc(),
|
||||||
r_target_ptr, r_eye_ptr, r_up_in_ptr,
|
.t4 = ralloc(),
|
||||||
r_tmp0, r_tmp1, r_tmp2, r_tmp3);
|
|
||||||
|
|
||||||
/* === ATOM 1: normalize fwd→uz === */
|
|
||||||
U2 src_offset = O_(ResolveLookAtScratch, fwd);
|
|
||||||
U2 dst_offset = O_(ResolveLookAtScratch, uz);
|
|
||||||
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
|
|
||||||
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
|
|
||||||
.scratch = R_ResolveScratch,
|
|
||||||
.src_ptr = R_T0,
|
|
||||||
.dst_ptr = R_T1,
|
|
||||||
.recip_est = R_T6,
|
|
||||||
.norm = R_T7,
|
|
||||||
.shift = R_V0,
|
|
||||||
.src_x = R_T2,
|
|
||||||
.t3 = R_T3,
|
|
||||||
.t4 = R_T5,
|
|
||||||
.t5 = R_V1,
|
|
||||||
});
|
});
|
||||||
|
regfile_reset(& rf);
|
||||||
|
|
||||||
/* === ATOM 2: cross uz×up_in→right === */
|
bundle->normalize_fwd_uz = smem.normalize_v3s4;
|
||||||
U4 r_a_2 = R_T0;
|
bundle->cross_to_right = smem.gte_cross_v3s4;
|
||||||
U4 r_b_2 = R_T1;
|
bundle->normalize_right_ux = smem.normalize_v3s4;
|
||||||
U4 r_c_2 = R_T2;
|
bundle->cross_to_up = smem.gte_cross_v3s4;
|
||||||
U4 r_d_2 = R_T3;
|
bundle->normalize_up_uy = smem.normalize_v3s4;
|
||||||
U4 r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
|
|
||||||
U4 r_g_2 = R_T6; /* a ptr = scratch+16 */
|
|
||||||
U4 r_h_2 = R_T7; /* b ptr = scratch+128 */
|
|
||||||
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
|
|
||||||
R_ResolveScratch,
|
|
||||||
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
|
|
||||||
|
|
||||||
/* === ATOM 3: normalize right→ux === */
|
bundle->pop_mv_trans = resolve_look_at__pop_mv_trans(& ab,
|
||||||
src_offset = O_(ResolveLookAtScratch, right);
|
RegUse_(resolve_look_at__pop_mv_trans){
|
||||||
dst_offset = O_(ResolveLookAtScratch, ux);
|
.look_at = ralloc(),
|
||||||
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
|
.eye = ralloc(),
|
||||||
src_offset, dst_offset, RegUse_(normalize_v3s4_proc){
|
.row = ralloc_v3(),
|
||||||
.scratch = R_ResolveScratch,
|
.t6 = ralloc(),
|
||||||
.src_ptr = R_T0,
|
.t7 = ralloc(),
|
||||||
.dst_ptr = R_T1,
|
.t8 = ralloc(),
|
||||||
.recip_est = R_T6,
|
|
||||||
.norm = R_T7,
|
|
||||||
.shift = R_V0,
|
|
||||||
.src_x = R_T2,
|
|
||||||
.t3 = R_T3,
|
|
||||||
.t4 = R_T5,
|
|
||||||
.t5 = R_V1,
|
|
||||||
});
|
});
|
||||||
|
|
||||||
/* === ATOM 4: cross uz×ux→up === */
|
|
||||||
U4 r_a_4 = R_T0;
|
|
||||||
U4 r_b_4 = R_T1;
|
|
||||||
U4 r_c_4 = R_T2;
|
|
||||||
U4 r_d_4 = R_T3;
|
|
||||||
U4 r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
|
|
||||||
U4 r_g_4 = R_T6; /* a ptr = scratch+16 */
|
|
||||||
U4 r_h_4 = R_T7; /* b ptr = scratch+48 */
|
|
||||||
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
|
|
||||||
R_ResolveScratch,
|
|
||||||
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
|
|
||||||
|
|
||||||
/* === ATOM 5: normalize up→uy === */
|
|
||||||
src_offset = O_(ResolveLookAtScratch, up);
|
|
||||||
dst_offset = O_(ResolveLookAtScratch, uy);
|
|
||||||
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
|
|
||||||
src_offset, dst_offset,
|
|
||||||
RegUse_(normalize_v3s4_proc){
|
|
||||||
.scratch = R_ResolveScratch,
|
|
||||||
.src_ptr = R_T0,
|
|
||||||
.dst_ptr = R_T1,
|
|
||||||
.recip_est = R_T6,
|
|
||||||
.norm = R_T7,
|
|
||||||
.shift = R_V0,
|
|
||||||
.src_x = R_T2,
|
|
||||||
.t3 = R_T3,
|
|
||||||
.t4 = R_T5,
|
|
||||||
.t5 = R_V1,
|
|
||||||
});
|
|
||||||
|
|
||||||
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
|
|
||||||
U4 r_look_at_6a = R_T0; /* tape pop → look_at* */
|
|
||||||
U4 r_scratch_6a = R_ResolveScratch;
|
|
||||||
U4 r_pux_6a = R_T1;
|
|
||||||
U4 r_puy_6a = R_T3;
|
|
||||||
U4 r_puz_6a = R_T5;
|
|
||||||
U4 r_tmp0_6a = R_T2;
|
|
||||||
U4 r_tmp1_6a = R_T6;
|
|
||||||
U4 r_tmp2_6a = R_V0;
|
|
||||||
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
|
|
||||||
r_look_at_6a, r_scratch_6a,
|
|
||||||
r_pux_6a, r_puy_6a, r_puz_6a,
|
|
||||||
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
|
|
||||||
|
|
||||||
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
|
|
||||||
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
|
|
||||||
* the matrix pointer (popped from tape). It does NOT need GPR
|
|
||||||
* assignment from us — it has its own internal GPR usage.
|
|
||||||
* We just take its address. */
|
|
||||||
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
|
|
||||||
|
|
||||||
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
|
|
||||||
* Uses mac_apply_matrix_lv component macro which internally uses
|
|
||||||
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
|
|
||||||
* for the mfc2/store. We pass our GPRs. */
|
|
||||||
U4 r_scratch_6b = R_ResolveScratch;
|
|
||||||
U4 r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
|
|
||||||
U4 r_look_at_6b = R_T0; /* tape pop → look_at* */
|
|
||||||
U4 r_tmp0_6b = R_T2;
|
|
||||||
U4 r_tmp1_6b = R_T3;
|
|
||||||
U4 r_tmp2_6b = R_T5;
|
|
||||||
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
|
|
||||||
r_scratch_6b, r_peye_6b, r_look_at_6b,
|
|
||||||
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
|
|
||||||
|
|
||||||
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
|
|
||||||
U4 r_look_at_6c = R_T0; /* tape pop → look_at* */
|
|
||||||
U4 r_scratch_6c = R_ResolveScratch;
|
|
||||||
U4 r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
|
|
||||||
U4 r_tmp0_6c = R_T2;
|
|
||||||
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
|
|
||||||
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c, R_T3, R_T4);
|
|
||||||
|
|
||||||
/* Sanity check: arena didn't overflow. */
|
/* Sanity check: arena didn't overflow. */
|
||||||
assert(ab.used <= ResolveLookAtArena_Size);
|
assert(ab.used <= ResolveLookAtArena_Size);
|
||||||
|
#undef ralloc
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update(). */
|
||||||
* The 7 chain atoms are pre-built at init time (resolve_look_at_init) and referenced by address via smem.resolve_look_at_atom_addrs[].
|
I_ void resolve_look_at(TapeBuilder_R tb
|
||||||
* Per-frame work: 7 tb_emit (atom pointer emissions) + 5 tb_data (C-side pointers for atom 0 + look_at for atom 6).
|
|
||||||
*
|
|
||||||
* Binds_ contract (the field-name labels are for human readability):
|
|
||||||
* Atom 0 input_and_sub target(4) eye(4) up_in(4) scratch_base(4) = 4 words
|
|
||||||
* Atoms 1-5 (no tape data — atom uses r_scratch + offset internally)
|
|
||||||
* Atom 6 populate_and_translate look_at(4) = 1 word
|
|
||||||
* ----
|
|
||||||
* 5 tb_data words total per frame.
|
|
||||||
*/
|
|
||||||
I_ void resolve_look_at(
|
|
||||||
TapeBuilder_R tb
|
|
||||||
, MT3_S2S4* look_at
|
, MT3_S2S4* look_at
|
||||||
, P3_S4* eye
|
, P3_S4* eye
|
||||||
, P3_S4* target
|
, P3_S4* target
|
||||||
, V3_S4* up_in
|
, V3_S4* up_in
|
||||||
){
|
){
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
|
/* Typed view of the scratchpad for field-address arithmetic. */
|
||||||
|
ResolveLookAtScratch* sp = C_scratch(ResolveLookAtScratch*);
|
||||||
|
AtomBundle_resolve_look_at_R bundle = C_(void*, smem.resolve_look_at_bundle);
|
||||||
|
|
||||||
|
tb_emit(tb, bundle->input_and_sub); {
|
||||||
tb_data(tb, u4_(target));
|
tb_data(tb, u4_(target));
|
||||||
tb_data(tb, u4_(eye));
|
tb_data(tb, u4_(eye));
|
||||||
tb_data(tb, u4_(up_in));
|
tb_data(tb, u4_(up_in));
|
||||||
tb_data(tb, u4_(smem.scratchpad));
|
|
||||||
}
|
}
|
||||||
|
tb_emit(tb, bundle->normalize_fwd_uz); {
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
|
tb_data(tb, u4_(O_(ResolveLookAtScratch, fwd) | (O_(ResolveLookAtScratch, uz) << 16)));
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
|
}
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
|
tb_emit(tb, bundle->cross_to_right); {
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
|
tb_data(tb, u4_(& sp->uz));
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
|
tb_data(tb, u4_(& sp->up_in));
|
||||||
|
tb_data(tb, u4_(& sp->right));
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
|
}
|
||||||
|
tb_emit(tb, bundle->normalize_right_ux); {
|
||||||
|
tb_data(tb, u4_(O_(ResolveLookAtScratch, right) | (O_(ResolveLookAtScratch, ux) << 16)));
|
||||||
|
}
|
||||||
|
tb_emit(tb, bundle->cross_to_up); {
|
||||||
|
tb_data(tb, u4_(& sp->uz));
|
||||||
|
tb_data(tb, u4_(& sp->ux));
|
||||||
|
tb_data(tb, u4_(& sp->up));
|
||||||
|
}
|
||||||
|
tb_emit(tb, bundle->normalize_up_uy); {
|
||||||
|
tb_data(tb, u4_(O_(ResolveLookAtScratch, up) | (O_(ResolveLookAtScratch, uy) << 16)));
|
||||||
|
}
|
||||||
|
tb_emit(tb, bundle->pop_mv_trans); {
|
||||||
tb_data(tb, u4_(look_at));
|
tb_data(tb, u4_(look_at));
|
||||||
}
|
}
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
|
|
||||||
tb_data(tb, u4_(look_at));
|
|
||||||
}
|
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
|
|
||||||
tb_data(tb, u4_(look_at));
|
|
||||||
}
|
|
||||||
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
|
|
||||||
// tb_data(tb, u4_(look_at));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
GCC_OPTIMIZATION_DISABLE
|
GCC_OPTIMIZATION_DISABLE
|
||||||
@@ -365,12 +289,6 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
gknown V3_S4_R acc = & smem.cube.accel;
|
gknown V3_S4_R acc = & smem.cube.accel;
|
||||||
add_v3s4(vel, acc[0]);
|
add_v3s4(vel, acc[0]);
|
||||||
add_v3s4_fp(pos, vel[0]);
|
add_v3s4_fp(pos, vel[0]);
|
||||||
// vel->x += acc->x;
|
|
||||||
// vel->y += acc->y;
|
|
||||||
// vel->z += acc->z;
|
|
||||||
// pos->x += vel->x;
|
|
||||||
// pos->y += vel->y;
|
|
||||||
// pos->z += vel->z;
|
|
||||||
|
|
||||||
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||||
|
|
||||||
@@ -403,9 +321,6 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
gte_matrix_set_rotation (& smem.tform_view);
|
gte_matrix_set_rotation (& smem.tform_view);
|
||||||
gte_matrix_set_translation(& smem.tform_view);
|
gte_matrix_set_translation(& smem.tform_view);
|
||||||
|
|
||||||
// gte_matrix_set_rotation (& smem.tform_world);
|
|
||||||
// gte_matrix_set_translation(& smem.tform_world);
|
|
||||||
|
|
||||||
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
U4 prim_cursor = prim_base + pa->used;
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
@@ -425,7 +340,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
tb_data(& tb, u4_(& pa->used));
|
tb_data(& tb, u4_(& pa->used));
|
||||||
tb_data(& tb, prim_base);
|
tb_data(& tb, prim_base);
|
||||||
}
|
}
|
||||||
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||||
|
|
||||||
// smem.cube.rot.y += 30;
|
// smem.cube.rot.y += 30;
|
||||||
}
|
}
|
||||||
@@ -467,7 +382,7 @@ void update(PrimitiveArena* pa, U4* ordering_buf)
|
|||||||
tb_data(& tb, u4_(& pa->used));
|
tb_data(& tb, u4_(& pa->used));
|
||||||
tb_data(& tb, prim_base);
|
tb_data(& tb, prim_base);
|
||||||
}
|
}
|
||||||
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
tape_run(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||||
|
|
||||||
// C-side state (pa->used) has already been updated by the tape!
|
// C-side state (pa->used) has already been updated by the tape!
|
||||||
// smem.floor.rot.y += 5;
|
// smem.floor.rot.y += 5;
|
||||||
@@ -519,8 +434,8 @@ int main(void)
|
|||||||
/* Direct BIOS: poll both ports during VBlank. */
|
/* Direct BIOS: poll both ports during VBlank. */
|
||||||
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||||
|
|
||||||
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
|
compile_init_atoms();
|
||||||
resolve_look_at_init();
|
compile_resolve_look_at();
|
||||||
|
|
||||||
/* Pinned registers for the GPU init atom. */
|
/* Pinned registers for the GPU init atom. */
|
||||||
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||||
@@ -541,4 +456,3 @@ int main(void)
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
GCC_OPTIMIZATION_ENABLE
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|
||||||
|
|||||||
+34
-2865
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,912 @@
|
|||||||
|
--- duffle_emit.lua — project_emission + decl finders.
|
||||||
|
local scan = require("duffle_scan")
|
||||||
|
local isa = require("duffle_isa")
|
||||||
|
local M = {}
|
||||||
|
for k, v in pairs(scan) do M[k] = v end
|
||||||
|
for k, v in pairs(isa) do M[k] = v end
|
||||||
|
|
||||||
|
-- Section 8: Cross-source component-body index + word-event expansion
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
--
|
||||||
|
-- Shared, memoized helpers: a single emitted-word event stream that every downstream pass reads from,
|
||||||
|
-- built once from the pre-tokenized bodies.
|
||||||
|
|
||||||
|
--- @class ComponentBodyEntry
|
||||||
|
--- @field body_tokens table -- pre-tokenized {{tok=string, rel=integer}, ...}
|
||||||
|
--- @field body_off integer -- byte offset of body[1] in `source`
|
||||||
|
--- @field line_of fun(pos:integer):integer -- byte-offset → 1-based line number in `source`
|
||||||
|
--- @field source string -- absolute path of the source containing the declaration
|
||||||
|
--- @field declaration integer -- 1-based line number of the MipsAtomComp_(ac_X) declaration
|
||||||
|
--- @field kind string -- "comp_bare" | "comp_proc"
|
||||||
|
|
||||||
|
-- The cross-source component-body index is owned by the corpus (`corpus.component_body_index`, populated by `passes/components.lua`).
|
||||||
|
-- Consumers (`passes/static_analysis.lua`, `passes/emission_model.lua`) read it directly; per-pass memoization helpers stay out of scope.
|
||||||
|
|
||||||
|
-- ASCII byte constants used by split_call_args (kept local to keep Section 8 self-contained).
|
||||||
|
local E_BYTE_OPEN_PAREN = 0x28
|
||||||
|
local E_BYTE_OPEN_BRACE = 0x7B
|
||||||
|
local E_BYTE_OPEN_BRACK = 0x5B
|
||||||
|
local E_BYTE_DQUOTE = 0x22
|
||||||
|
local E_BYTE_SQUOTE = 0x27
|
||||||
|
local E_BYTE_COMMA = 0x2C
|
||||||
|
|
||||||
|
-- Map an open-delimiter byte to its matching close string for read_balanced.
|
||||||
|
local E_OPEN_CLOSE = {
|
||||||
|
[E_BYTE_OPEN_PAREN] = ")",
|
||||||
|
[E_BYTE_OPEN_BRACE] = "}",
|
||||||
|
[E_BYTE_OPEN_BRACK] = "]",
|
||||||
|
}
|
||||||
|
|
||||||
|
--- Split the INSIDE of a `f(...)` call on top-level commas.
|
||||||
|
--- Honors nested parens / braces / brackets and skips strings / comments.
|
||||||
|
--- Returns a list of trimmed argument strings in source order.
|
||||||
|
--- (Mirrors split_top_level_commas but for paren-body args; intentionally distinct so a caller's brace-body split isn't confused with an arg list.)
|
||||||
|
--- @param inner string
|
||||||
|
--- @return string[]
|
||||||
|
local function split_call_args(inner)
|
||||||
|
local args = {}
|
||||||
|
if not inner or inner == "" then return args end
|
||||||
|
local pos = 1
|
||||||
|
local len = #inner
|
||||||
|
local start = 1
|
||||||
|
while pos <= len do
|
||||||
|
local c = inner:byte(pos)
|
||||||
|
local close = E_OPEN_CLOSE[c]
|
||||||
|
if close then
|
||||||
|
local _, after = M.read_balanced(inner, string.char(c), close, pos)
|
||||||
|
pos = after
|
||||||
|
elseif c == E_BYTE_DQUOTE or c == E_BYTE_SQUOTE then
|
||||||
|
pos = M.skip_str_or_cmt(inner, pos)
|
||||||
|
elseif c == E_BYTE_COMMA then
|
||||||
|
args[#args + 1] = M.trim(inner:sub(start, pos - 1))
|
||||||
|
start = pos + 1
|
||||||
|
pos = pos + 1
|
||||||
|
else
|
||||||
|
pos = pos + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if start <= len then args[#args + 1] = M.trim(inner:sub(start, len)) end
|
||||||
|
return args
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Extract the leading identifier + top-level args list from a token string.
|
||||||
|
--- Returns (ident, args). For tokens without a `(...)` call, args is `{}`.
|
||||||
|
--- @param tok string
|
||||||
|
--- @return string, string[]
|
||||||
|
local function token_ident_and_args(tok)
|
||||||
|
local ident, after = M.read_ident(tok, 1)
|
||||||
|
if not ident then return "?", {} end
|
||||||
|
local paren_pos = M.skip_ws_and_cmt(tok, after)
|
||||||
|
if tok:sub(paren_pos, paren_pos) ~= "(" then return ident, {} end
|
||||||
|
local inner = M.read_parens(tok, paren_pos)
|
||||||
|
if not inner then return ident, {} end
|
||||||
|
return ident, split_call_args(inner)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- The macro-name prefix that marks a `mac_X(...)` component invocation.
|
||||||
|
local E_MAC_PREFIX = "mac_"
|
||||||
|
local E_MAC_PREFIX_LEN = 4
|
||||||
|
|
||||||
|
--- Expand a body entry into the flat sequence of emitted machine-word events.
|
||||||
|
---
|
||||||
|
--- Semantics (one event per emitted machine word):
|
||||||
|
--- * Direct one-word encoders `load_word`, `add_ui`, `nop`, `gte_lw`, ...: One event with `ident` = leading ident, `args` = parsed top-level args.
|
||||||
|
--- * `nop2` (2-word pseudo-instruction): Two events, both with `ident = "nop"` so the recognized "this slot is a no-op" semantic is visible to downstream analyses.
|
||||||
|
--- * Any other N-word token in `word_counts`: N events sharing the same `ident` + `args` so useful CPU words retire slots in the cycle budget.
|
||||||
|
--- * Known `mac_X(...)` calls: Recursively expand the indexed component body, including nested components. Every event from the expansion carries:
|
||||||
|
--- - `source` / `line` = the COMPONENT'S source path + the line of the token within the component body (i.e. "definition site").
|
||||||
|
--- - `call_source` / `call_line` = the ROOT atom's source path + call-site line, PRESERVED across recursion so nested events still point at the original root.
|
||||||
|
--- * Unknown `mac_X` (not in `component_index`): fall back to `word_counts[ident]` if present; otherwise emit one opaque event so the cycle budget accounts for the word.
|
||||||
|
--- * Marker Tokens (`atom_label(...)` / `atom_offset(...)`): Zero events (they are pure metaprogram hints).
|
||||||
|
---
|
||||||
|
--- Cycle protection: a per-expansion `visiting` set tracks components currently on the expansion stack;
|
||||||
|
--- a re-entry produces a deterministic `{kind = "cycle", ...}` error and aborts that branch (does NOT hang, does NOT recurse).
|
||||||
|
---
|
||||||
|
--- Pure: reads `body_entry` / `component_index` / `word_counts`. Memoization is the caller's responsibility.
|
||||||
|
--- Callers wanting `word_events` / `word_event_errors` precomputed for many atoms should memoize them per atom.
|
||||||
|
--- @param body_entry table -- `{body_tokens, body_off, line_of, source, declaration}` (declaration = root atom's atom.line)
|
||||||
|
--- @param component_index table -- the bare-name → ComponentBodyEntry map from M.get_component_body_index
|
||||||
|
--- @param word_counts table -- macro name → emitted-word count (from `ctx.shared.word_counts`)
|
||||||
|
--- @return WordEvent[], WordEventError[]
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Section 11: project_emission (per-atom emission projection)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
--
|
||||||
|
-- Per-atom emission projection is owned by `passes/emission_model.lua`.
|
||||||
|
-- The projection is built from the root atom body only; invocation ancestry recursively expands nested components.
|
||||||
|
-- The items stream is the single ordered source of truth; `word_events` and `markers` are dense views over it.
|
||||||
|
--
|
||||||
|
-- The helper below operates on a body string (not a body_entry) so the pass can call it without depending on the older SourceScan / body_off conventions.
|
||||||
|
-- component_index argument is reserved for recursive component expansion.
|
||||||
|
-- word_counts table is authored-metadata + current-component count table.
|
||||||
|
|
||||||
|
--- @class EmissionProjection
|
||||||
|
--- @field items table[] -- Ordered stream of word|label|offset|invoke_begin|invoke_end
|
||||||
|
--- @field word_events table[] -- Dense view of items where kind == "word"
|
||||||
|
--- @field markers table[] -- Dense view of items where kind == "label"|"offset"
|
||||||
|
--- @field invocations InvocationRecord[] -- dense view of items where kind == "invoke_begin"|"invoke_end"
|
||||||
|
--- @field errors table[] -- Token-resolution failures surfaced without fail-loud
|
||||||
|
--- @field warnings table[] -- Opaque warnings (e.g. unknown uncounted macro)
|
||||||
|
|
||||||
|
--- @class InvocationRecord
|
||||||
|
--- Lives at `atom.paths.invocations[*]`. Constructed once at the single invocation-construction site
|
||||||
|
--- (`emit_invoke_begin` inside `_project_emission_inner`); `invoke_begin` / `invoke_end` markers in the items stream share the same `id`.
|
||||||
|
--- @field id integer -- 1-based, monotonic per-atom invocation id (0 is reserved for "no open invocation")
|
||||||
|
--- @field parent_id integer -- 0 for the outermost (root) call; otherwise the id of the immediately enclosing invocation
|
||||||
|
--- @field kind string -- "comp_bare" | "comp_proc" (component form that triggered the expansion)
|
||||||
|
--- @field component_name string -- Bare component name without the `mac_` prefix
|
||||||
|
--- @field call_text string -- Immediate `mac_X(...)` token text (or root call text for the outermost entry)
|
||||||
|
--- @field root_call_text string -- IMMUTABLE outermost `mac_X(...)` token text for every word emitted in this call's expansion
|
||||||
|
--- @field call_path string -- Source path of the call site (root atom source for direct calls, component source for nested expansions)
|
||||||
|
--- @field call_line integer -- Source line of the call site
|
||||||
|
--- @field def_path string -- Source path of the component definition
|
||||||
|
--- @field def_line integer -- Source line of the component declaration
|
||||||
|
--- @field start_pos integer -- 0-based emitted-word position of the FIRST word inside this invocation (the value of `word_idx` AT `emit_invoke_begin` time, BEFORE the first word is emitted). Words emitted inside this invocation occupy `start_pos..start_pos+#body_lines-1` (inclusive, 0-based). Downstream DWARF/provenance consumers MUST read this; do NOT reconstruct it from `start_word` (which is the 1-based items index including `invoke_begin`/`invoke_end` markers).
|
||||||
|
--- @field end_pos integer -- 0-based position of the LAST word inside this invocation (set by `emit_invoke_end` to `word_idx - 1` AFTER all body words are emitted).
|
||||||
|
--- @field start_word integer -- 1-based items index of the `invoke_begin` item
|
||||||
|
--- @field end_word integer -- 1-based items index of the `invoke_end` item (set by `emit_invoke_end`)
|
||||||
|
--- @field word_count integer -- Number of `word` items emitted between `start_word` and `end_word` (inclusive)
|
||||||
|
--- @field debug_skip boolean -- `debug_skip` stamp; true iff `corpus.components[name].debug_skip` is true at construction. Always boolean (never `nil`).
|
||||||
|
--- @field errors table[] -- Per-invocation construction errors (cycle / count_mismatch); does not include pass-level errors
|
||||||
|
|
||||||
|
-- Internal recursive walker. The items stream holds every emitted event in order; `word_events`, `markers`,
|
||||||
|
-- `invocations`, `errors`, `warnings` are dense views / side outputs appended alongside.
|
||||||
|
--
|
||||||
|
-- Output rules:
|
||||||
|
-- * `word` items record: `invocation_ids` (innermost last) and `outermost_invocation_id` (0 if no invocation is open).
|
||||||
|
-- * `invoke_begin` / `invoke_end` items are zero-width at the current word index; the same `word_index` is recorded on both.
|
||||||
|
-- * `root_call_text` is the outermost `mac_X(...)` token text for every word emitted inside a component expansion;
|
||||||
|
-- it is `nil` for direct words emitted from the root atom body.
|
||||||
|
-- * `call_text` is the IMMEDIATE top-level token spelling for the word (for nested words this is the inner `mac_X(...)` token;
|
||||||
|
-- for direct words it is the trimmed encoder token).
|
||||||
|
-- * `def_path` / `def_line` are the definition site of the current body (component source for nested words; root atom source for direct words, filled in by the pass caller).
|
||||||
|
-- * Unknown uncounted macros emit one opaque word + one warning. Unknown metadata-backed macros (entry in `word_counts`) emit the declared word count, no warning.
|
||||||
|
-- * Cycle detection uses an active DFS stack (`visiting`); a cycle appends a construction error to BOTH the projection errors and the cycle invocation's own errors,
|
||||||
|
-- then breaks out without recursing (the cycle entry still receives an invocation ID + paired `invoke_begin` / `invoke_end` items, so the boundary invariant is preserved).
|
||||||
|
-- * Component declared-count mismatch (declared vs. measured) is a construction error (kind = "count_mismatch"); recorded on the invocation record and pass-level errors list.
|
||||||
|
-- * Final boundary check: if any invocation is still open at end of walk, surface a "unbalanced" construction error.
|
||||||
|
local function _project_emission_inner(root_body_entry, ctx_table)
|
||||||
|
local items = {}
|
||||||
|
local word_events = {}
|
||||||
|
local markers = {}
|
||||||
|
local invocations = {}
|
||||||
|
local errors = {}
|
||||||
|
local warnings = {}
|
||||||
|
|
||||||
|
local word_idx = 0
|
||||||
|
local invocation_stack = {} -- stack of currently-open invocation records
|
||||||
|
local next_inv_id = 0
|
||||||
|
|
||||||
|
local reg_use_schema = ctx_table.reg_use_schema
|
||||||
|
local reg_use_param = ctx_table.reg_use_param
|
||||||
|
local atom_name = ctx_table.atom_name
|
||||||
|
|
||||||
|
local slot_readonly = {}
|
||||||
|
if reg_use_schema then
|
||||||
|
for _, slot in ipairs(reg_use_schema.slots or {}) do
|
||||||
|
slot_readonly[slot.name] = slot.readonly == true
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local function apply_sub(sub_map, operand)
|
||||||
|
if not (sub_map and type(operand) == "string") then return operand end
|
||||||
|
if sub_map[operand] then return sub_map[operand] end
|
||||||
|
local dot = operand:find(".", 1, true)
|
||||||
|
if dot then
|
||||||
|
local head = operand:sub(1, dot - 1)
|
||||||
|
local mapped = sub_map[head]
|
||||||
|
if type(mapped) == "string" then
|
||||||
|
return mapped .. operand:sub(dot)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return operand
|
||||||
|
end
|
||||||
|
|
||||||
|
local function resolve_gpr_key(operand)
|
||||||
|
if type(operand) ~= "string" then return nil end
|
||||||
|
if operand:sub(1, 2) == "R_" then return operand end
|
||||||
|
if not (reg_use_schema and reg_use_param) then return nil end
|
||||||
|
local prefix = reg_use_param .. "."
|
||||||
|
if operand:sub(1, #prefix) ~= prefix then return nil end
|
||||||
|
local member_path = operand:sub(#prefix + 1)
|
||||||
|
local slot = reg_use_schema.alias_to_slot[member_path]
|
||||||
|
if not slot then return nil, member_path end
|
||||||
|
return "reguse:" .. atom_name .. ":" .. slot, nil, slot
|
||||||
|
end
|
||||||
|
|
||||||
|
local function open_invocation_ids_snapshot()
|
||||||
|
local ids = {}
|
||||||
|
for _, inv in ipairs(invocation_stack) do
|
||||||
|
ids[#ids + 1] = inv.id
|
||||||
|
end
|
||||||
|
return ids
|
||||||
|
end
|
||||||
|
|
||||||
|
local function emit_word(encoder, args, line, word_call_text,
|
||||||
|
def_source_now, def_line_now,
|
||||||
|
immediate_call_text, root_call_text_w, sub_map)
|
||||||
|
local inv_ids = open_invocation_ids_snapshot()
|
||||||
|
local outermost = inv_ids[1] or 0
|
||||||
|
-- For words emitted at the root atom body, `immediate_call_text` is nil and the walker's `word_call_text` (the word's own token, e.g. "nop") becomes the effective call_text.
|
||||||
|
-- For words emitted inside a component expansion, `immediate_call_text` is the immediate outer `mac_X(...)` token text;
|
||||||
|
-- The call that triggered the body expansion we're currently walking.
|
||||||
|
local eff_call_text = immediate_call_text or word_call_text
|
||||||
|
local eff_root_call_text = root_call_text_w
|
||||||
|
local gpr_keys = nil
|
||||||
|
if reg_use_schema or sub_map then
|
||||||
|
gpr_keys = {}
|
||||||
|
for pos, arg in ipairs(args or {}) do
|
||||||
|
local effective = apply_sub(sub_map, arg)
|
||||||
|
local key, unresolved, slot = resolve_gpr_key(effective)
|
||||||
|
gpr_keys[pos] = key
|
||||||
|
if unresolved then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
kind = "reguse_unresolved",
|
||||||
|
line = line,
|
||||||
|
msg = string.format("RegUse operand %q does not resolve in schema %q",
|
||||||
|
effective, (reg_use_schema and reg_use_schema.name) or "?"),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
if key and slot and slot_readonly[slot] then
|
||||||
|
local row = M.instr(encoder)
|
||||||
|
if row and row.writes then
|
||||||
|
for _, wpos in ipairs(row.writes) do
|
||||||
|
if wpos == pos then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
kind = "reguse_const_write",
|
||||||
|
line = line,
|
||||||
|
msg = string.format("RegUse slot %q is Reg const; %s writes it",
|
||||||
|
slot, encoder),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if not reg_use_schema then
|
||||||
|
gpr_keys = nil
|
||||||
|
end
|
||||||
|
items[#items + 1] = {
|
||||||
|
kind = "word",
|
||||||
|
encoder = encoder,
|
||||||
|
args = args,
|
||||||
|
i = word_idx,
|
||||||
|
word_count = 1,
|
||||||
|
line = line,
|
||||||
|
call_text = eff_call_text,
|
||||||
|
root_call_text = eff_root_call_text,
|
||||||
|
invocation_ids = inv_ids,
|
||||||
|
outermost_invocation_id = outermost,
|
||||||
|
gpr_keys = gpr_keys,
|
||||||
|
}
|
||||||
|
word_events[#word_events + 1] = {
|
||||||
|
i = word_idx,
|
||||||
|
encoder = encoder,
|
||||||
|
args = args,
|
||||||
|
def_path = def_source_now or "",
|
||||||
|
def_line = def_line_now or 0,
|
||||||
|
call_text = eff_call_text,
|
||||||
|
root_call_text = eff_root_call_text,
|
||||||
|
invocation_ids = inv_ids,
|
||||||
|
outermost_invocation_id = outermost,
|
||||||
|
word_count = 1,
|
||||||
|
gpr_keys = gpr_keys,
|
||||||
|
}
|
||||||
|
word_idx = word_idx + 1
|
||||||
|
end
|
||||||
|
|
||||||
|
local function emit_marker(kind, name, target, line,
|
||||||
|
immediate_call_text, root_call_text_w,
|
||||||
|
consuming_encoder, consuming_arg_pos)
|
||||||
|
local inv_ids = open_invocation_ids_snapshot()
|
||||||
|
local outermost = inv_ids[1] or 0
|
||||||
|
-- Markers carry the open invocation stack snapshot. `call_text` / `root_call_text` belong to words, not markers — markers are zero-width and skip per-word call-site attribution.
|
||||||
|
-- `consuming_encoder` + `consuming_arg_pos` carry the surrounding control-transfer instruction context
|
||||||
|
-- (e.g. `branch_le_zero` consuming its 3rd argument, or `jump` / `call_addr` consuming their only argument).
|
||||||
|
-- `passes/offsets.lua` reads these to dispatch per-consuming-instruction offset encoding.
|
||||||
|
-- nil for top-level markers (where the marker is the entire token — no surrounding consuming instruction).
|
||||||
|
local it = {
|
||||||
|
kind = kind,
|
||||||
|
name = name,
|
||||||
|
line = line,
|
||||||
|
word_index = word_idx,
|
||||||
|
invocation_ids = inv_ids,
|
||||||
|
outermost_invocation_id = outermost,
|
||||||
|
}
|
||||||
|
if target ~= nil then it.target = target end
|
||||||
|
if consuming_encoder then it.consuming_encoder = consuming_encoder end
|
||||||
|
if consuming_arg_pos then it.consuming_arg_pos = consuming_arg_pos end
|
||||||
|
items[#items + 1] = it
|
||||||
|
markers[#markers + 1] = {
|
||||||
|
kind = kind,
|
||||||
|
name = name,
|
||||||
|
line = line,
|
||||||
|
word_index = word_idx,
|
||||||
|
target = target,
|
||||||
|
consuming_encoder = consuming_encoder,
|
||||||
|
consuming_arg_pos = consuming_arg_pos,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Count top-level commas in `tok` between position `from_pos` (inclusive) and `to_pos` (exclusive).
|
||||||
|
-- Tracks paren depth so commas inside nested () don't count. Skips string literals + comments.
|
||||||
|
-- Used by `emit_embedded_markers` to compute `consuming_arg_pos` for each embedded marker.
|
||||||
|
local function count_top_level_commas(tok, from_pos, to_pos)
|
||||||
|
local depth = 0
|
||||||
|
local count = 0
|
||||||
|
local i = from_pos
|
||||||
|
while i < to_pos do
|
||||||
|
local c = tok:sub(i, i)
|
||||||
|
if c == "'" or c == '"' then
|
||||||
|
local next_pos = M.skip_str_or_cmt(tok, i)
|
||||||
|
i = (next_pos > i) and next_pos or (i + 1)
|
||||||
|
elseif c == "/" and tok:sub(i + 1, i + 1) == "/" then
|
||||||
|
-- line comment: skip to end of line
|
||||||
|
local nl = tok:find("\n", i, true)
|
||||||
|
i = (nl and nl + 1) or (#tok + 1)
|
||||||
|
elseif c == "/" and tok:sub(i + 1, i + 1) == "*" then
|
||||||
|
-- block comment: skip to matching */
|
||||||
|
local close = tok:find("*/", i + 2, true)
|
||||||
|
i = (close and close + 2) or (#tok + 1)
|
||||||
|
elseif c == "(" then
|
||||||
|
depth = depth + 1
|
||||||
|
i = i + 1
|
||||||
|
elseif c == ")" then
|
||||||
|
depth = depth - 1
|
||||||
|
i = i + 1
|
||||||
|
elseif c == "," and depth == 0 then
|
||||||
|
count = count + 1
|
||||||
|
i = i + 1
|
||||||
|
else
|
||||||
|
i = i + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return count
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Find the position of the consuming instruction's open paren (the `(` that starts the consuming instruction's argument list).
|
||||||
|
-- Returns nil if the token's leading text isn't an ident followed by `(` (e.g. the ident is at the start of a non-instruction token).
|
||||||
|
local function find_consuming_paren(tok)
|
||||||
|
local i = 1
|
||||||
|
while i <= #tok do
|
||||||
|
local c = tok:sub(i, i)
|
||||||
|
if c == "(" then return i end
|
||||||
|
if not c:match("[%w_]") and c ~= " " then return nil end
|
||||||
|
i = i + 1
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
local function emit_embedded_markers(tok, tok_line, consuming_encoder)
|
||||||
|
-- When called with a non-nil `consuming_encoder`, the marker is nested inside that instruction's argument list.
|
||||||
|
-- We compute each marker's arg position by counting top-level commas between the consuming instruction's `(` and the marker's start.
|
||||||
|
local consuming_paren = nil
|
||||||
|
if consuming_encoder then consuming_paren = find_consuming_paren(tok) end
|
||||||
|
local pos = 1
|
||||||
|
while pos <= #tok do
|
||||||
|
-- Trim leading whitespace and comments before each scan.
|
||||||
|
pos = M.skip_ws_and_cmt(tok, pos)
|
||||||
|
if pos > #tok then break end
|
||||||
|
local ident, after = M.read_ident(tok, pos)
|
||||||
|
if not ident then
|
||||||
|
-- Not an ident: token is a string or comment; skip or one-step.
|
||||||
|
local next_pos = M.skip_str_or_cmt(tok, pos)
|
||||||
|
pos = (next_pos > pos) and next_pos or (pos + 1)
|
||||||
|
goto continue_loop
|
||||||
|
end
|
||||||
|
if M.DELAY_MARKERS[ident] then
|
||||||
|
local arg_pos = nil
|
||||||
|
if consuming_encoder and consuming_paren then
|
||||||
|
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
|
||||||
|
end
|
||||||
|
emit_marker("delay", ident, nil, tok_line, nil, nil, consuming_encoder, arg_pos)
|
||||||
|
pos = after
|
||||||
|
goto continue_loop
|
||||||
|
end
|
||||||
|
if ident ~= "atom_label" and ident ~= "atom_offset" then
|
||||||
|
-- Ordinary ident; nothing to emit, step past the ident only.
|
||||||
|
pos = after
|
||||||
|
goto continue_loop
|
||||||
|
end
|
||||||
|
-- Marker ident: parse the (...) arguments.
|
||||||
|
local open = M.skip_ws_and_cmt(tok, after)
|
||||||
|
local inner, after_paren = M.read_parens(tok, open)
|
||||||
|
if not inner then
|
||||||
|
-- (...) Unreadable: fall back to non-marker behavior.
|
||||||
|
pos = after
|
||||||
|
goto continue_loop
|
||||||
|
end
|
||||||
|
-- Commit: label takes 1 arg, offset takes 2.
|
||||||
|
-- For embedded markers, propagate the consuming_encoder + the marker's arg position
|
||||||
|
-- (1-based) so `passes/offsets.lua` can dispatch per-consuming-instruction offset encoding.
|
||||||
|
-- Top-level markers (no consuming_encoder) get nil for both — the offsets pass treats
|
||||||
|
-- them as branch-equivalent for backward compatibility.
|
||||||
|
local arg_pos = nil
|
||||||
|
if consuming_encoder and consuming_paren then
|
||||||
|
arg_pos = count_top_level_commas(tok, consuming_paren + 1, pos) + 1
|
||||||
|
end
|
||||||
|
local args = split_call_args(inner)
|
||||||
|
if ident == "atom_label" then emit_marker("label", args[1] or "", nil, tok_line, nil, nil, consuming_encoder, arg_pos)
|
||||||
|
else emit_marker("offset", args[1] or "", args[2] or "", tok_line, nil, nil, consuming_encoder, arg_pos)
|
||||||
|
end
|
||||||
|
pos = after_paren
|
||||||
|
::continue_loop::
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local function emit_invoke_begin(inv_kind, component_name, call_text,
|
||||||
|
root_call_text, call_path, call_line)
|
||||||
|
next_inv_id = next_inv_id + 1
|
||||||
|
-- Invocation-level debug_skip stamp: Emission pass owns `atom.paths.invocations[*].debug_skip`.
|
||||||
|
-- The stamp is resolved from the `corpus.components[name]` registry (passed in via `ctx_table.components` by `emission_model.run`),
|
||||||
|
-- Unmarked components stamp `false` (not `nil`) so consumers can dispatch on the boolean without nil checks.
|
||||||
|
--
|
||||||
|
-- The walker has already found the component body in `ctx_table.component_index[component_name]`, so the matching entry MUST exist in `ctx_table.components[component_name]`
|
||||||
|
-- (both registries are populated from the same source by the components pass).
|
||||||
|
-- A missing entry is a corpus-plumbing bug; we fail loudly here rather than silently stamp `false` and mask the regression.
|
||||||
|
local components = ctx_table.components
|
||||||
|
local component_def = components and components[component_name] or nil
|
||||||
|
if not component_def then
|
||||||
|
error("duffle.emit_invoke_begin: component " .. string.format("%q", component_name)
|
||||||
|
.. " is present in `component_index` (the walker matched a `mac_" .. component_name .. "()` call) but absent from `components` (the canonical corpus.components registry). "
|
||||||
|
.. "This is a corpus-plumbing bug — the components pass must populate corpus.components[name] for every component it puts in corpus.component_body_index[name]. "
|
||||||
|
.. "The emission pass refuses to silently stamp `debug_skip = false` for a missing registry entry."
|
||||||
|
, 0
|
||||||
|
)
|
||||||
|
end
|
||||||
|
local debug_skip_stamp = component_def.debug_skip == true
|
||||||
|
local inv = {
|
||||||
|
id = next_inv_id,
|
||||||
|
parent_id = 0, -- patched below by caller
|
||||||
|
kind = inv_kind,
|
||||||
|
component_name = component_name,
|
||||||
|
call_text = call_text,
|
||||||
|
root_call_text = root_call_text,
|
||||||
|
call_path = call_path,
|
||||||
|
call_line = call_line,
|
||||||
|
def_path = nil, -- patched below after component lookup
|
||||||
|
def_line = nil,
|
||||||
|
-- 0-based emitted-word position. `word_idx` is the monotonic 0-based counter of `word` items emitted so far in this walk —
|
||||||
|
-- BEFORE this invocation's first word is emitted, it equals the position of the first word inside the invocation.
|
||||||
|
-- `start_word` (1-based items index of `invoke_begin`) is kept for items-walking consumers (Annotation pass bounds checks),
|
||||||
|
-- but DWARF / provenance rows MUST read `start_pos` because those rows are 1-based over the dense `word_events` stream (which has no `invoke_begin` items).
|
||||||
|
start_pos = word_idx,
|
||||||
|
start_word = #items + 1, -- 1-based items index of invoke_begin
|
||||||
|
end_pos = nil, -- patched by emit_invoke_end
|
||||||
|
end_word = nil, -- patched by emit_invoke_end
|
||||||
|
word_count = 0,
|
||||||
|
debug_skip = debug_skip_stamp,
|
||||||
|
errors = {},
|
||||||
|
}
|
||||||
|
invocations[#invocations + 1] = inv
|
||||||
|
items [#items + 1] = {
|
||||||
|
kind = "invoke_begin",
|
||||||
|
invocation_id = inv.id,
|
||||||
|
word_index = word_idx,
|
||||||
|
invocation_ids = open_invocation_ids_snapshot(),
|
||||||
|
}
|
||||||
|
invocation_stack[#invocation_stack + 1] = inv
|
||||||
|
return inv
|
||||||
|
end
|
||||||
|
|
||||||
|
local function emit_invoke_end(inv)
|
||||||
|
-- 0-based emitted-word position of the LAST word inside this invocation.
|
||||||
|
-- After the last body word was emitted, `word_idx` was incremented past it, so `word_idx - 1` is the 0-based position of the last word.
|
||||||
|
inv.end_pos = word_idx - 1
|
||||||
|
inv.end_word = #items + 1 -- 1-based items index of invoke_end
|
||||||
|
items[#items + 1] = {
|
||||||
|
kind = "invoke_end",
|
||||||
|
invocation_id = inv.id,
|
||||||
|
word_index = word_idx,
|
||||||
|
invocation_ids = open_invocation_ids_snapshot(),
|
||||||
|
}
|
||||||
|
for i = #invocation_stack, 1, -1 do
|
||||||
|
if invocation_stack[i] == inv then
|
||||||
|
table.remove(invocation_stack, i)
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Resolve the per-token word count. If unresolved, surface ONE warning
|
||||||
|
-- and fall back to 1 opaque word so the cycle budget still accounts for the slot.
|
||||||
|
local function resolve_count(ident, tok_line)
|
||||||
|
local wc = ctx_table.word_counts
|
||||||
|
if wc and wc[ident] then return wc[ident] end
|
||||||
|
local canon = M.gte_canon(ident)
|
||||||
|
if canon ~= ident and wc and wc[canon] then return wc[canon] end
|
||||||
|
warnings[#warnings + 1] = {
|
||||||
|
kind = "uncounted",
|
||||||
|
line = tok_line,
|
||||||
|
msg = string.format("project_emission: opaque word emitted for %q (no entry in word_counts or component_index)",
|
||||||
|
ident),
|
||||||
|
}
|
||||||
|
return 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Recursive walker: walk one body entry, possibly descending into components.
|
||||||
|
-- walk_parent_inv_id: Invocation ID of the enclosing call (0 for the root call).
|
||||||
|
-- walk_root_call_text: Outermost `mac_X(...)` token text (preserved across recursion).
|
||||||
|
-- walk_immediate_call_text: IMMEDIATE outer `mac_X(...)` token text for words emitted in this body — nil for the root atom body.
|
||||||
|
-- Two trackers are propagated as separate parameters so words deep inside nested expansions correctly identify both their immediate call site and the outermost call site.
|
||||||
|
local function walk_body_entry(body_entry, walk_parent_inv_id,
|
||||||
|
walk_root_call_text, walk_immediate_call_text)
|
||||||
|
local tokens = body_entry.body_tokens or {}
|
||||||
|
local body_off = body_entry.body_off or 0
|
||||||
|
local line_of = body_entry.line_of or M.LineIndex("")
|
||||||
|
local def_source = body_entry.source or ""
|
||||||
|
local def_line = body_entry.declaration or 0
|
||||||
|
local sub_map = body_entry.sub_map
|
||||||
|
-- Per-token dispatch: each matched branch returns; only the fall-through
|
||||||
|
-- "opaque word" emit handles direct encoders + mac_X-without-component.
|
||||||
|
local function process_token(bt)
|
||||||
|
local tok = M.trim(bt.tok or "")
|
||||||
|
if tok == "" then return end
|
||||||
|
local ident, after = M.read_ident(tok, 1)
|
||||||
|
if not ident then ident = "?" end
|
||||||
|
local _, args = token_ident_and_args(tok)
|
||||||
|
local tok_line = line_of(body_off + bt.rel) or 0
|
||||||
|
if M.DELAY_MARKERS[ident] then
|
||||||
|
emit_marker("delay", ident, nil, tok_line)
|
||||||
|
local rest = M.trim(tok:sub(after or (#tok + 1)))
|
||||||
|
if rest ~= "" then
|
||||||
|
process_token({ tok = rest, rel = bt.rel })
|
||||||
|
end
|
||||||
|
return
|
||||||
|
end
|
||||||
|
-- embedded markers live only in non-marker tokens.
|
||||||
|
-- Pass `ident` as the consuming instruction so `emit_embedded_markers` can compute each marker's arg position + record the consuming_encoder for the offsets pass.
|
||||||
|
-- Canonicalize `jump_rel` to `branch_equal` (its preprocessor-expanded form) so the `consuming_encoder` metadata in marker records is canonical.
|
||||||
|
-- `jump_rel`: unconditional jump alias from `code/duffle/mips.h`.
|
||||||
|
local consuming_encoder_for_markers = (ident == "jump_rel") and "branch_equal" or ident
|
||||||
|
if ident ~= "atom_label" and ident ~= "atom_offset" then
|
||||||
|
emit_embedded_markers(tok, tok_line, consuming_encoder_for_markers)
|
||||||
|
end
|
||||||
|
-- atom_label / atom_offset: terminal markers, no further descent.
|
||||||
|
-- Top-level markers (the marker IS the entire token) have no consuming instruction;
|
||||||
|
-- nil for both `consuming_encoder` and `consuming_arg_pos`.
|
||||||
|
-- The offsets pass treats these as branch-equivalent for backward compatibility.
|
||||||
|
-- TODO(Ed): Review this don't want legacy cruft here..
|
||||||
|
if ident == "atom_label" then emit_marker("label", args[1] or "", nil, tok_line); return
|
||||||
|
elseif ident == "atom_offset" then emit_marker("offset", args[1] or "", args[2] or "", tok_line); return
|
||||||
|
end
|
||||||
|
if ident:sub(1, 4) == "mac_" then
|
||||||
|
local bare = ident:sub(5)
|
||||||
|
local comp = ctx_table.component_index[bare]
|
||||||
|
if comp then
|
||||||
|
local invocation_root_call_text = walk_root_call_text or tok
|
||||||
|
if ctx_table.visiting[bare] then
|
||||||
|
-- Cycle: still allocate inv_id, emit zero-width begin/end, record the cycle error; do NOT recurse.
|
||||||
|
local inv = emit_invoke_begin(comp.kind or "comp_bare", bare, tok, invocation_root_call_text, def_source, tok_line)
|
||||||
|
inv.parent_id = walk_parent_inv_id
|
||||||
|
inv.call_text = tok
|
||||||
|
local err = {
|
||||||
|
kind = "cycle",
|
||||||
|
msg = string.format("project_emission: component cycle detected: %q", bare),
|
||||||
|
source = def_source,
|
||||||
|
line = tok_line,
|
||||||
|
}
|
||||||
|
inv.errors[#inv.errors + 1] = err
|
||||||
|
errors [#errors + 1] = err
|
||||||
|
emit_invoke_end(inv)
|
||||||
|
return
|
||||||
|
end
|
||||||
|
-- First visit: descend + count + count_mismatch-check below.
|
||||||
|
ctx_table.visiting[bare] = true
|
||||||
|
local inv = emit_invoke_begin(comp.kind or "comp_bare", bare, tok, invocation_root_call_text, def_source, tok_line)
|
||||||
|
inv.parent_id = walk_parent_inv_id
|
||||||
|
inv.call_text = tok
|
||||||
|
inv.def_path = comp.source
|
||||||
|
inv.def_line = comp.declaration
|
||||||
|
-- Propagate trackers into the recursive walk:
|
||||||
|
-- immediate_call_text = this call's tok (the IMMEDIATE outer call for words emitted in this body)
|
||||||
|
-- root_call_text = the OUTERMOST call (immutable across the recursion)
|
||||||
|
local formal_names = ctx_table.component_index[bare]
|
||||||
|
and ctx_table.component_index[bare].arg_names
|
||||||
|
local child_map = nil
|
||||||
|
if formal_names then
|
||||||
|
child_map = {}
|
||||||
|
for i, fname in ipairs(formal_names) do
|
||||||
|
child_map[fname] = apply_sub(sub_map, args[i])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
walk_body_entry({
|
||||||
|
body_tokens = comp.body_tokens or {},
|
||||||
|
body_off = comp.body_off or 0,
|
||||||
|
line_of = comp.line_of,
|
||||||
|
source = comp.source,
|
||||||
|
declaration = comp.declaration,
|
||||||
|
sub_map = child_map,
|
||||||
|
},
|
||||||
|
inv.id,
|
||||||
|
invocation_root_call_text,
|
||||||
|
tok)
|
||||||
|
ctx_table.visiting[bare] = nil
|
||||||
|
emit_invoke_end(inv)
|
||||||
|
-- Count `word` items inside [start_word, end_word].
|
||||||
|
local wc_inside = 0
|
||||||
|
for i = inv.start_word, inv.end_word do
|
||||||
|
local it = items[i]
|
||||||
|
if it and it.kind == "word" then
|
||||||
|
wc_inside = wc_inside + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
inv.word_count = wc_inside
|
||||||
|
-- count_mismatch is a construction error: word_counts["mac_X"] is the declared count populated by the components pass;
|
||||||
|
-- We compare against the measured word count.
|
||||||
|
local declared = ctx_table.word_counts["mac_" .. bare]
|
||||||
|
if declared and wc_inside ~= declared then
|
||||||
|
local err = {
|
||||||
|
kind = "count_mismatch",
|
||||||
|
msg = string.format("project_emission: mac_%s declared=%d measured=%d", bare, declared, wc_inside),
|
||||||
|
source = def_source,
|
||||||
|
line = tok_line,
|
||||||
|
}
|
||||||
|
inv.errors[#inv.errors + 1] = err
|
||||||
|
errors [#errors + 1] = err
|
||||||
|
end
|
||||||
|
return
|
||||||
|
end
|
||||||
|
-- mac_X NOT in component_index: fall through to opaque emit.
|
||||||
|
end
|
||||||
|
-- Direct encoder, or mac_X-without-component: resolve count + emit n words.
|
||||||
|
-- Resolve_count may emit a warning if the count is unresolved.
|
||||||
|
local n = resolve_count(ident, tok_line)
|
||||||
|
local out_ident = (ident == "nop2") and "nop" or ident
|
||||||
|
for _ = 1, n do
|
||||||
|
emit_word(out_ident, args, tok_line, tok, def_source, def_line, walk_immediate_call_text, walk_root_call_text, sub_map)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
for _, bt in ipairs(tokens) do
|
||||||
|
process_token(bt)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Initialize the per-walk mutable context.
|
||||||
|
-- `visiting` is the active DFS component stack; `root_call_path` / `root_call_line` are preserved across recursion so nested words always point at the
|
||||||
|
-- ORIGINAL root atom call site.
|
||||||
|
ctx_table.visiting = ctx_table.visiting or {}
|
||||||
|
ctx_table.root_call_path = ctx_table.root_call_path or ""
|
||||||
|
ctx_table.root_call_line = ctx_table.root_call_line or 0
|
||||||
|
|
||||||
|
-- Walk first; the pass caller stamps the root call site for direct words after the projection returns.
|
||||||
|
-- For nested words the def_path / def_line already point at the component source and MUST be preserved (the stamping helper checks for that).
|
||||||
|
walk_body_entry(root_body_entry, 0, nil, nil)
|
||||||
|
|
||||||
|
-- Boundary check: every invoke_begin must have a matching invoke_end.
|
||||||
|
-- If anything is still open, surface a hard error.
|
||||||
|
if #invocation_stack > 0 then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
kind = "unbalanced",
|
||||||
|
msg = string.format("project_emission: invocation boundaries not balanced (%d unclosed invocation(s) at end of walk)", #invocation_stack),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
return {
|
||||||
|
items = items,
|
||||||
|
word_events = word_events,
|
||||||
|
markers = markers,
|
||||||
|
invocations = invocations,
|
||||||
|
errors = errors,
|
||||||
|
warnings = warnings,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Project a body string into the per-atom emission projection.
|
||||||
|
---
|
||||||
|
--- Semantics:
|
||||||
|
--- * Direct one-word tokens (`nop`, `add_ui`, ...): one `word` item, encoder = ident, word_count = 1.
|
||||||
|
--- * Metadata-backed N-word tokens (`nop2`, `mask_upper`, ...): N `word` items, all sharing the same encoder + word_count = 1.
|
||||||
|
--- `nop2` is normalized to encoder `nop` (per the spec).
|
||||||
|
--- * `atom_label(F)` markers: one `label` item with `name = "F"`, `word_index = current word_idx`; zero-width (does NOT advance word_idx).
|
||||||
|
--- * `atom_offset(B, T)` markers: one `offset` item with `name = "B"`, `target = "T"`, `word_index = current word_idx`; zero-width.
|
||||||
|
--- * Delay markers (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`): one `delay` item; zero-width. The following encoder is the next token.
|
||||||
|
--- * `mac_X(...)` calls: emit `invoke_begin` (zero-width), recurse into the component body, emit `invoke_end` (zero-width).
|
||||||
|
--- The component body's words land between the begin/end pair; one invocation record is allocated per call (monotonic ID per atom).
|
||||||
|
--- * Unknown uncounted macros emit 1 opaque word + one warning per occurrence.
|
||||||
|
--- * Tokens whose count cannot be resolved (e.g. `mac_unknown` not in word_counts and not in component_index) surface one
|
||||||
|
--- warning; cycle + count-mismatch + boundary violations are construction errors on `pass.errors`.
|
||||||
|
---
|
||||||
|
--- Every emitted `word` carries: `i` (0-based word index), `encoder`, `args` (top-level args), `def_path`, `def_line`,
|
||||||
|
--- `call_text` (the immediate token spelling), `root_call_text` (outermost `mac_X(...)` text), `word_count` (always 1),
|
||||||
|
--- `invocation_ids` (innermost last), `outermost_invocation_id`.
|
||||||
|
--- Markers carry: `kind`, `name`, `line`, `word_index`, `target` (only for offset kind), plus `invocation_ids` / `outermost_invocation_id`
|
||||||
|
--- for the open invocation stack at that word.
|
||||||
|
---
|
||||||
|
--- @param body_text string -- the raw atom body string
|
||||||
|
--- @param component_index table -- bare-name → component record (corpus.component_body_index)
|
||||||
|
--- @param word_counts table -- macro name → emitted word count
|
||||||
|
--- @param components table -- bare-name → component definition (corpus.components); REQUIRED — consumed at the invocation-construction site to stamp
|
||||||
|
--- `invocation.debug_skip`. A missing or non-table `components` raises a fail-loud error rather than silently falling back.
|
||||||
|
--- @return EmissionProjection
|
||||||
|
function M.project_emission(body_text, component_index, word_counts, components, reg_use_ctx)
|
||||||
|
-- The recursive walk delegates to `_project_emission_inner` so component bodies (which arrive as
|
||||||
|
-- `{body_tokens, body_off, line_of, source, declaration}` records from `corpus.component_body_index`)
|
||||||
|
-- re-enter the same walker with the same shared output state.
|
||||||
|
--
|
||||||
|
-- The walker is body-relative: it builds `line_of` from `body_text` and stamps body-relative line numbers (1..N)
|
||||||
|
-- into `item.line` and `invocation.call_line`. `passes/emission_model.lua::stamp_root_provenance` performs the single
|
||||||
|
-- conversion from body-relative to physical source line at the close site, using the source's `line_of` closure that
|
||||||
|
-- the pass forwarded. One owner of the line state.
|
||||||
|
if type(components) ~= "table" then
|
||||||
|
error("duffle.project_emission: `components` is required "
|
||||||
|
.. "(bare-name -> component definition, e.g. corpus.components); "
|
||||||
|
.. "got " .. type(components) .. ". "
|
||||||
|
.. "The emission pass MUST forward the corpus registry "
|
||||||
|
.. "so the invocation-construction site can stamp `debug_skip` "
|
||||||
|
.. "without a second pass, source parse, or parallel lookup.",
|
||||||
|
0)
|
||||||
|
end
|
||||||
|
|
||||||
|
if type(body_text) ~= "string" or body_text == "" then
|
||||||
|
-- Empty body: still return a valid (empty) projection.
|
||||||
|
return {
|
||||||
|
items = {},
|
||||||
|
word_events = {},
|
||||||
|
markers = {},
|
||||||
|
invocations = {},
|
||||||
|
errors = {},
|
||||||
|
warnings = {},
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
local tokens = M.tokenize_body(body_text)
|
||||||
|
return _project_emission_inner({
|
||||||
|
body_tokens = tokens,
|
||||||
|
body_off = 0,
|
||||||
|
line_of = M.LineIndex(body_text),
|
||||||
|
source = "",
|
||||||
|
declaration = 0,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
component_index = component_index or {},
|
||||||
|
word_counts = word_counts or {},
|
||||||
|
components = components,
|
||||||
|
reg_use_schema = reg_use_ctx and reg_use_ctx.reg_use_schema,
|
||||||
|
reg_use_param = reg_use_ctx and reg_use_ctx.reg_use_param,
|
||||||
|
atom_name = reg_use_ctx and reg_use_ctx.atom_name,
|
||||||
|
schema_name = reg_use_ctx and reg_use_ctx.schema_name,
|
||||||
|
})
|
||||||
|
end
|
||||||
|
|
||||||
|
-------------------------------------------------------------------------------
|
||||||
|
-- find_function_decl_for — backward walk for MipsAtomComp_Proc_ name extraction.
|
||||||
|
--
|
||||||
|
-- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name is derived from the preceding
|
||||||
|
-- `FI_ Slice_MipsCode ac_X(args)` function declaration. This function walks backward from `before_pos` to find it.
|
||||||
|
--
|
||||||
|
-- Returns (raw_name, args_inner) or (nil, nil).
|
||||||
|
-- raw_name — e.g. "ac_load_word_imm"
|
||||||
|
-- args_inner — e.g. "AtomBuilder_R ab, Reg dst, U4 imm"
|
||||||
|
--
|
||||||
|
-- The walk finds the LAST "Slice_MipsCode" before before_pos, then skips whitespace + qualifiers
|
||||||
|
-- (FI_, atom_dbg_skip, comments) until it finds an ident followed by "(".
|
||||||
|
-- That ident is the function name; the parens contents are the args.
|
||||||
|
-------------------------------------------------------------------------------
|
||||||
|
function M.find_function_decl_for(source, before_pos, slice_mips_code_len)
|
||||||
|
local search_pos = 1
|
||||||
|
local last_match = nil
|
||||||
|
while true do
|
||||||
|
local found = source:find("Slice_MipsCode", search_pos, true)
|
||||||
|
if not found or found >= before_pos then break end
|
||||||
|
last_match = found
|
||||||
|
search_pos = found + slice_mips_code_len
|
||||||
|
end
|
||||||
|
if not last_match then return nil, nil end
|
||||||
|
|
||||||
|
local pos = last_match + slice_mips_code_len
|
||||||
|
while pos < before_pos do
|
||||||
|
-- skip whitespace
|
||||||
|
while pos <= #source do
|
||||||
|
local c = source:sub(pos, pos)
|
||||||
|
if c == " " or c == "\t" or c == "\n" or c == "\r" then
|
||||||
|
pos = pos + 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if pos > #source then break end
|
||||||
|
-- skip line comments
|
||||||
|
if source:sub(pos, pos + 1) == "//" then
|
||||||
|
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
|
||||||
|
pos = pos + 1
|
||||||
|
goto continue
|
||||||
|
end
|
||||||
|
-- skip block comments
|
||||||
|
if source:sub(pos, pos + 1) == "/*" then
|
||||||
|
local close = source:find("*/", pos + 2, true)
|
||||||
|
if not close then break end
|
||||||
|
pos = close + 2
|
||||||
|
goto continue
|
||||||
|
end
|
||||||
|
-- try to read an ident
|
||||||
|
local ident, ident_end = M.read_ident(source, pos)
|
||||||
|
if not ident then break end
|
||||||
|
-- check if the next non-ws char after ident is "("
|
||||||
|
local next_pos = M.skip_ws_and_cmt(source, ident_end)
|
||||||
|
if source:sub(next_pos, next_pos) == "(" then
|
||||||
|
local inner = M.read_parens(source, next_pos)
|
||||||
|
if inner then
|
||||||
|
return ident, inner
|
||||||
|
end
|
||||||
|
end
|
||||||
|
-- ident not followed by "(" — it's a qualifier (FI_, atom_dbg_skip, etc); skip it
|
||||||
|
pos = ident_end
|
||||||
|
::continue::
|
||||||
|
end
|
||||||
|
return nil, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-------------------------------------------------------------------------------
|
||||||
|
-- find_atom_proc_decl_for — backward walk for MipsAtom_Proc_ name extraction.
|
||||||
|
--
|
||||||
|
-- The atom name is the preceding `MipsAtom* ident(args)` function ident.
|
||||||
|
-- This function walks backward from `before_pos` to find it.
|
||||||
|
--
|
||||||
|
-- Returns (raw_name, args_inner, func_ident, after_paren) or (nil, nil).
|
||||||
|
-- raw_name — the function ident as written
|
||||||
|
-- args_inner — e.g. "AtomArena_R aa, U4 r_scratch, ..."
|
||||||
|
-- after_paren — source position after the function `)`
|
||||||
|
--
|
||||||
|
-- The walk finds the LAST "MipsAtom*" before before_pos, then skips whitespace + qualifiers (internal, I_, FI_, comments)
|
||||||
|
-- until it finds an ident followed by "(".
|
||||||
|
-------------------------------------------------------------------------------
|
||||||
|
function M.find_atom_proc_decl_for(source, before_pos, mips_atom_ptr_len)
|
||||||
|
local search_pos = 1
|
||||||
|
local last_match = nil
|
||||||
|
while true do
|
||||||
|
-- plain=true: "*" is literal, no escaping needed
|
||||||
|
local found = source:find("MipsAtom*", search_pos, true)
|
||||||
|
if not found or found >= before_pos then break end
|
||||||
|
last_match = found
|
||||||
|
search_pos = found + mips_atom_ptr_len
|
||||||
|
end
|
||||||
|
if not last_match then return nil, nil end
|
||||||
|
|
||||||
|
local pos = last_match + mips_atom_ptr_len
|
||||||
|
while pos < before_pos do
|
||||||
|
-- skip whitespace
|
||||||
|
while pos <= #source do
|
||||||
|
local c = source:sub(pos, pos)
|
||||||
|
if c == " " or c == "\t" or c == "\n" or c == "\r" then
|
||||||
|
pos = pos + 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if pos > #source then break end
|
||||||
|
-- skip line comments
|
||||||
|
if source:sub(pos, pos + 1) == "//" then
|
||||||
|
while pos <= #source and source:sub(pos, pos) ~= "\n" do pos = pos + 1 end
|
||||||
|
pos = pos + 1
|
||||||
|
goto continue
|
||||||
|
end
|
||||||
|
-- skip block comments
|
||||||
|
if source:sub(pos, pos + 1) == "/*" then
|
||||||
|
local close = source:find("*/", pos + 2, true)
|
||||||
|
if not close then break end
|
||||||
|
pos = close + 2
|
||||||
|
goto continue
|
||||||
|
end
|
||||||
|
-- try to read an ident
|
||||||
|
local ident, ident_end = M.read_ident(source, pos)
|
||||||
|
if not ident then break end
|
||||||
|
-- check if the next non-ws char after ident is "("
|
||||||
|
local next_pos = M.skip_ws_and_cmt(source, ident_end)
|
||||||
|
if source:sub(next_pos, next_pos) == "(" then
|
||||||
|
local inner, after_paren = M.read_parens(source, next_pos)
|
||||||
|
if inner then
|
||||||
|
return ident, inner, ident, after_paren
|
||||||
|
end
|
||||||
|
end
|
||||||
|
-- ident not followed by "(" — it's a qualifier; skip it
|
||||||
|
pos = ident_end
|
||||||
|
::continue::
|
||||||
|
end
|
||||||
|
return nil, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -0,0 +1,725 @@
|
|||||||
|
--- duffle_isa.lua — encoder / GTE / hardware tables.
|
||||||
|
local M = {}
|
||||||
|
|
||||||
|
-- Section 7: domain tables
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- atom_info sub-calls: atom_bind, atom_reads, atom_writes, atom_view, atom_reg_types, atom_ctx, atom_phase.
|
||||||
|
M.TAPE_ATOM_MACROS = {
|
||||||
|
["atom_info"] = { kind = "info", binds = false },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Empty C macros that prefix the next encoder. Zero words.
|
||||||
|
-- BdSlot_ nop is one nop word. The marker is not the BD instruction.
|
||||||
|
M.DELAY_MARKERS = {
|
||||||
|
["GteDelay_"] = true,
|
||||||
|
["LdSlot_"] = true,
|
||||||
|
["BdSlot_"] = true,
|
||||||
|
["DmaSlot_"] = true,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- One row per encoder. Old table names are load-time views (build_isa_views).
|
||||||
|
M.INSTRUCTION = {
|
||||||
|
["BdSlot_"] = { cycles = 0, kind = "marker", },
|
||||||
|
["LdSlot_"] = { cycles = 0, kind = "marker", },
|
||||||
|
["add_s"] = { cycles = 1, kind = "alu", },
|
||||||
|
["add_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, },}, },
|
||||||
|
["add_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["add_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "add_u", sources = { 1, 2 }, }, },
|
||||||
|
["add_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
||||||
|
["add_ui_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, signed = true, width = 16, }, }, value = { dest = 1, immediate = 2, op = "add_ui", source = 1, }, },
|
||||||
|
["and"] = { cycles = 1, kind = "alu", },
|
||||||
|
["and_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "and_i", source = 2, }, },
|
||||||
|
["and_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["atom_bind"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_info"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_label"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_offset"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_reads"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["atom_writes"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["branch_equal"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["branch_ge_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_gt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_le_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_lt_zero"] = { cycles = 2, kind = "branch", reads = { 1 }, writes = {}, imm = { { arg = 2, signed = true, width = 16, }, }, },
|
||||||
|
["branch_ne"] = { cycles = 2, kind = "branch", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["call_addr"] = { cycles = 2, kind = "call", reads = {}, writes = { 1 }, },
|
||||||
|
["call_reg"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
||||||
|
["div_s"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["div_u"] = { cycles = 35, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["gte_load_v0"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_load_v0v1v2"] = { cycles = 6, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_load_v1"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_load_v2"] = { cycles = 2, kind = "cop2_xfer", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_lw"] = { cycles = 1, kind = "load", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_lwc2"] = { cycles = 1, kind = "load", },
|
||||||
|
["gte_mv_from_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
||||||
|
["gte_mv_from_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = { 1 }, },
|
||||||
|
["gte_mv_to_ctrl_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
||||||
|
["gte_mv_to_data_r"] = { cycles = 1, kind = "cop2_xfer", reads = { 1 }, writes = {}, },
|
||||||
|
["gte_stotz"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
||||||
|
["gte_stsxy3"] = { cycles = 1, kind = "cop2_xfer", reads = {}, writes = {}, },
|
||||||
|
["gte_sw"] = { cycles = 1, kind = "store", reads = { 2 }, writes = {}, },
|
||||||
|
["gte_swc2"] = { cycles = 1, kind = "store", },
|
||||||
|
["jump"] = { cycles = 2, kind = "jump", reads = {}, writes = {}, },
|
||||||
|
["jump_link"] = { cycles = 2, kind = "call", reads = { 1 }, writes = { 2 }, },
|
||||||
|
["jump_reg"] = { cycles = 2, kind = "jump", reads = { 1 }, writes = {}, suppress_arg1 = { R_AtomJmp = "fixed mac_yield handshake", }, },
|
||||||
|
["jump_rel"] = { cycles = 2, kind = "branch", delay_slot = true, },
|
||||||
|
["li_s"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, immediate = 3, op = "add_ui", source = 2, }, },
|
||||||
|
["load_byte"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_byte_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_half"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_half_u"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["load_imm"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["load_ui"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["load_upper_i"] = { cycles = 1, kind = "alu", reads = {}, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "load_upper_i", }, },
|
||||||
|
["load_word"] = { cycles = 1, kind = "load", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["mac_yield"] = { cycles = 0, kind = "marker", reads = {}, writes = {}, },
|
||||||
|
["mask_upper"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||||
|
["mov_from_high"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["mov_from_low"] = { cycles = 2, kind = "alu", reads = {}, writes = { 1 }, },
|
||||||
|
["mov_to_high"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
||||||
|
["mov_to_low"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = {}, },
|
||||||
|
["mult_s"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["mult_u"] = { cycles = 12, kind = "alu", reads = { 1, 2 }, writes = {}, },
|
||||||
|
["nop"] = { cycles = 1, kind = "nop", reads = {}, writes = {}, },
|
||||||
|
["nop2"] = { cycles = 2, kind = "nop", reads = {}, writes = {}, },
|
||||||
|
["nor_u"] = { cycles = 1, kind = "alu", },
|
||||||
|
["or_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "or_i", source = 2, }, },
|
||||||
|
["or_i_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 16, }, }, value = { dest = 1, immediate = 2, op = "or_i", source = 1, }, },
|
||||||
|
["or_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["or_u_self"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, value = { dest = 1, op = "or", sources = { 1, 2 }, }, },
|
||||||
|
["set_lt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["set_lt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||||
|
["set_lt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["set_lt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, },
|
||||||
|
["shift_aright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["shift_aright_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["shift_lleft"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["shift_lleft_self"] = { cycles = 1, kind = "alu", reads = { 1 }, writes = { 1 }, imm = { { arg = 2, width = 5, }, }, value = { dest = 1, immediate = 2, op = "shift_lleft", source = 1, }, },
|
||||||
|
["shift_lleft_var"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["shift_lright"] = { cycles = 1, kind = "alu", reads = { 2 }, writes = { 1 }, imm = { { arg = 3, width = 5, }, }, },
|
||||||
|
["slt_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["slt_si"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["slt_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["slt_ui"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, signed = true, width = 16,}, }, },
|
||||||
|
["store_byte"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["store_half"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["store_word"] = { cycles = 1, kind = "store", reads = { 1, 2 }, writes = {}, imm = { { arg = 3, signed = true, width = 16, }, }, },
|
||||||
|
["sub_s"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["sub_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
["sys_mov_from_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = {}, writes = { 1 }, },
|
||||||
|
["sys_mov_to_cop0"] = { cycles = 1, kind = "cop0_xfer", reads = { 1 }, writes = {}, },
|
||||||
|
["xor_i"] = { cycles = 1, kind = "alu", reads = { 1, 2 }, writes = { 1 }, imm = { { arg = 3, width = 16, }, }, value = { dest = 1, immediate = 3, op = "xor_i", source = 2, }, },
|
||||||
|
["xor_u"] = { cycles = 1, kind = "alu", reads = { 2, 3 }, writes = { 1 }, },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- One row per GTE command. Alias cycle numbers live here, not on INSTRUCTION.
|
||||||
|
M.GTE_COMMAND = {
|
||||||
|
["gte_cmdw_avsz3"] = {
|
||||||
|
aliases = { "gte_avg_sort_z3", "gte_avsz3", "gte_cmdw_avg_sort_z3" },
|
||||||
|
cycles = 5,
|
||||||
|
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF3" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_avsz4"] = {
|
||||||
|
aliases = { "gte_avg_sort_z4", "gte_avsz4", "gte_cmdw_avg_sort_z4" },
|
||||||
|
cycles = 6,
|
||||||
|
inputs = { "C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3", "gte_cr_ZSF4" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_gpf"] = {
|
||||||
|
aliases = {},
|
||||||
|
cycles = 5,
|
||||||
|
inputs = { "C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_MAC1", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC2", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC3", role = "mac_result", },
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_MAC1", required = 4, },
|
||||||
|
{ register = "C2_MAC2", required = 4, },
|
||||||
|
{ register = "C2_MAC3", required = 4, },
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_mvmva"] = {
|
||||||
|
aliases = {},
|
||||||
|
cycles = 8,
|
||||||
|
inputs = {
|
||||||
|
"C2_VXY0", "C2_VZ0",
|
||||||
|
"C2_VXY1", "C2_VZ1",
|
||||||
|
"C2_VXY2", "C2_VZ2",
|
||||||
|
"C2_IR1", "C2_IR2", "C2_IR3",
|
||||||
|
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||||
|
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||||
|
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||||
|
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ"
|
||||||
|
},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_nclip"] = {
|
||||||
|
aliases = { "gte_nclip" },
|
||||||
|
cycles = 8,
|
||||||
|
inputs = { "C2_SXY0", "C2_SXY1", "C2_SXY2" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_SZ3", role = "mac_result", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_SZ3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_op"] = {
|
||||||
|
aliases = { "gte_cmdw_outer_product", "gte_cmdw_wedge" },
|
||||||
|
cycles = 6,
|
||||||
|
inputs = {},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_rtps"] = {
|
||||||
|
aliases = { "gte_cmdw_rotate_translate_perspective_single", "gte_rtps" },
|
||||||
|
cycles = 15,
|
||||||
|
inputs = {
|
||||||
|
"C2_VXY0", "C2_VZ0",
|
||||||
|
"C2_VXY1", "C2_VZ1",
|
||||||
|
"C2_VXY2", "C2_VZ2",
|
||||||
|
"C2_RGB", "C2_OTZ",
|
||||||
|
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||||
|
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||||
|
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||||
|
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||||
|
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||||
|
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
||||||
|
"gte_cr_OFX", "gte_cr_OFY",
|
||||||
|
"gte_cr_H",
|
||||||
|
"gte_cr_DQA", "gte_cr_DQB"
|
||||||
|
},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
||||||
|
{ register = "C2_SZ2", role = "latest_screen_z", },
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
{ register = "C2_IR0", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_SXY2", required = 4, },
|
||||||
|
{ register = "C2_SZ2", required = 4, },
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
{ register = "C2_IR0", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_rtpt"] = {
|
||||||
|
aliases = { "gte_cmdw_rotate_translate_perspective_triple", "gte_rtpt" },
|
||||||
|
cycles = 23,
|
||||||
|
inputs = {
|
||||||
|
"C2_VXY0", "C2_VZ0",
|
||||||
|
"C2_VXY1", "C2_VZ1",
|
||||||
|
"C2_VXY2", "C2_VZ2",
|
||||||
|
"C2_RGB", "C2_OTZ",
|
||||||
|
"C2_IR0", "C2_IR1", "C2_IR2", "C2_IR3",
|
||||||
|
"C2_SZ0", "C2_SZ1", "C2_SZ2", "C2_SZ3",
|
||||||
|
"gte_cr_RT11", "gte_cr_RT12", "gte_cr_RT13",
|
||||||
|
"gte_cr_RT21", "gte_cr_RT22", "gte_cr_RT23",
|
||||||
|
"gte_cr_RT31", "gte_cr_RT32", "gte_cr_RT33",
|
||||||
|
"gte_cr_TRX", "gte_cr_TRY", "gte_cr_TRZ",
|
||||||
|
"gte_cr_OFX", "gte_cr_OFY",
|
||||||
|
"gte_cr_H",
|
||||||
|
"gte_cr_DQA", "gte_cr_DQB"
|
||||||
|
},
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_SXY0", role = "screen_xy[0]", },
|
||||||
|
{ register = "C2_SXY1", role = "screen_xy[1]", },
|
||||||
|
{ register = "C2_SXY2", role = "latest_screen_xy", },
|
||||||
|
{ register = "C2_SZ3", role = "latest_screen_z", },
|
||||||
|
{ register = "C2_OTZ", role = "otz", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_SXY0", required = 4, },
|
||||||
|
{ register = "C2_SXY1", required = 4, },
|
||||||
|
{ register = "C2_SXY2", required = 4, },
|
||||||
|
{ register = "C2_SZ3", required = 4, },
|
||||||
|
{ register = "C2_OTZ", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
["gte_cmdw_sqr"] = {
|
||||||
|
aliases = {},
|
||||||
|
cycles = 5,
|
||||||
|
inputs = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||||
|
outputs = {
|
||||||
|
{ register = "C2_MAC1", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC2", role = "mac_result", },
|
||||||
|
{ register = "C2_MAC3", role = "mac_result", },
|
||||||
|
{ register = "C2_IR1", role = "latest_color", },
|
||||||
|
{ register = "C2_IR2", role = "latest_color", },
|
||||||
|
{ register = "C2_IR3", role = "latest_color", },
|
||||||
|
},
|
||||||
|
latch = {
|
||||||
|
{ register = "C2_MAC1", required = 4, },
|
||||||
|
{ register = "C2_MAC2", required = 4, },
|
||||||
|
{ register = "C2_MAC3", required = 4, },
|
||||||
|
{ register = "C2_IR1", required = 4, },
|
||||||
|
{ register = "C2_IR2", required = 4, },
|
||||||
|
{ register = "C2_IR3", required = 4, },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
function M.instr (ident) return M.INSTRUCTION [ident] end
|
||||||
|
function M.gte_canon(ident) return M.ALIAS_TO_CANONICAL [ident] or ident end
|
||||||
|
function M.gte (ident) return M.GTE_COMMAND[M.gte_canon(ident)] end
|
||||||
|
|
||||||
|
local function build_isa_views()
|
||||||
|
M.ALIAS_TO_CANONICAL = {}
|
||||||
|
for canon, row in pairs(M.GTE_COMMAND) do
|
||||||
|
M.ALIAS_TO_CANONICAL[canon] = canon
|
||||||
|
for _, alias in ipairs(row.aliases or {}) do
|
||||||
|
M.ALIAS_TO_CANONICAL[alias] = canon
|
||||||
|
end
|
||||||
|
end
|
||||||
|
M.INSTRUCTION_LATENCY = {}
|
||||||
|
M.INSTRUCTION_GPR_EFFECTS = {}
|
||||||
|
M.IMMEDIATE_FIELD_WIDTHS = {}
|
||||||
|
M.GPR_VALUE_RULES = {}
|
||||||
|
M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES = {}
|
||||||
|
for name, row in pairs(M.INSTRUCTION) do
|
||||||
|
M.INSTRUCTION_LATENCY[name] = row.cycles
|
||||||
|
if row.reads or row.writes then
|
||||||
|
M.INSTRUCTION_GPR_EFFECTS[name] = {
|
||||||
|
reads = row.reads or {},
|
||||||
|
writes = row.writes or {},
|
||||||
|
}
|
||||||
|
end
|
||||||
|
if row.imm then M.IMMEDIATE_FIELD_WIDTHS[name] = row.imm end
|
||||||
|
if row.value then M.GPR_VALUE_RULES [name] = row.value end
|
||||||
|
if (row.kind == "branch" or row.kind == "jump" or row.kind == "call")
|
||||||
|
and row.delay_slot ~= false then
|
||||||
|
M.CONTROL_TRANSFER_DELAY_SLOT_POLICIES[name] = {
|
||||||
|
family = row.kind,
|
||||||
|
suppress_arg1 = row.suppress_arg1,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
M.GTE_COMMAND_ALIASES = {}
|
||||||
|
M.GTE_COMMAND_INPUTS = {}
|
||||||
|
M.GTE_COMMAND_OUTPUTS = {}
|
||||||
|
M.GTE_COMMAND_LATCH_WINDOWS = {}
|
||||||
|
for canon, row in pairs(M.GTE_COMMAND) do
|
||||||
|
M.GTE_COMMAND_ALIASES [canon] = canon
|
||||||
|
M.INSTRUCTION_LATENCY [canon] = row.cycles
|
||||||
|
M.INSTRUCTION_GPR_EFFECTS[canon] = { reads = {}, writes = {} }
|
||||||
|
for _, alias in ipairs(row.aliases or {}) do
|
||||||
|
M.GTE_COMMAND_ALIASES [alias] = canon
|
||||||
|
M.INSTRUCTION_LATENCY [alias] = row.cycles
|
||||||
|
M.INSTRUCTION_GPR_EFFECTS[alias] = { reads = {}, writes = {} }
|
||||||
|
end
|
||||||
|
M.GTE_COMMAND_INPUTS [canon] = row.inputs
|
||||||
|
M.GTE_COMMAND_OUTPUTS [canon] = row.outputs
|
||||||
|
M.GTE_COMMAND_LATCH_WINDOWS[canon] = row.latch
|
||||||
|
end
|
||||||
|
end
|
||||||
|
build_isa_views()
|
||||||
|
|
||||||
|
|
||||||
|
--- GTE control-register alias groups.
|
||||||
|
--- Aliases within a group write to the same C2 control-register slot (the HW double-maps some C2 slots across multiple PSX SDK / libgte conventions).
|
||||||
|
--- Aliases across groups write to distinct C2 slots.
|
||||||
|
---
|
||||||
|
--- Cross-alias writes inside one atom body, or across the wave-context boundary, silently clobber each other.
|
||||||
|
--- The `check_gte_cr_alias_writes` check warns about each pair per source. See `docs/gte_reference.md` §"Control-register alias table"
|
||||||
|
--- for the HW rationale and the libgte outer-product convention.
|
||||||
|
M.GTE_CR_ALIAS_GROUPS = {
|
||||||
|
{ 24, { "gte_cr_RBK", "gte_cr_OFX" } }, -- background R vs screen offset X
|
||||||
|
{ 25, { "gte_cr_GBK", "gte_cr_OFY" } }, -- background G vs screen offset Y
|
||||||
|
{ 26, { "gte_cr_BBK", "gte_cr_H" } }, -- background B vs projection plane distance H
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Packed RT slots named by the gte.h packed-slot comment. First must be written before second.
|
||||||
|
M.GTE_PACKED_SLOT_RELATIONS = {
|
||||||
|
{ slot = 2, first = "gte_cr_RT13", second = "gte_cr_RT22" },
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Operand-class table for the COP2->GPR load-delay check.
|
||||||
|
-- Maps each emitting-token ident to the set of GPR operand positions it reads.
|
||||||
|
-- Covers the current encoder vocabulary (`code/duffle/mips.h` + `code/duffle/gte.h`); add rows here as new encoders land.
|
||||||
|
--
|
||||||
|
-- Semantics:
|
||||||
|
-- * A "GPR operand position" is the textual slot in the macro's argument list, 1-based; e.g. `load_word(rt, base, off)` has positional operands 1 (rt), 2 (base), 3 (off).
|
||||||
|
-- The table reads operands 1 + 2 + 3 to find what GPRs the macro touches.
|
||||||
|
-- * The check tracks one entry per destination GPR per MFC2 / CFC2 event.
|
||||||
|
-- A subsequent event counts as a "use" iff any of its read operand positions reference that destination GPR's ident (e.g. `R_T0`).
|
||||||
|
-- * Branch delay slots are out of scope (MIPS control-flow; tracked separately).
|
||||||
|
M.OPERAND_READ_POSITIONS = {
|
||||||
|
-- CPU ALU with one or two GPR operands. Reads every GPR operand.
|
||||||
|
["add_ui"] = {1, 2},
|
||||||
|
["li_s"] = {1, 2}, -- rt (write), imm16 (immediate)
|
||||||
|
["add_ui_self"] = {1},
|
||||||
|
["add_si"] = {1, 2},
|
||||||
|
["add_u"] = {1, 2, 3},
|
||||||
|
["add_u_self"] = {1, 2},
|
||||||
|
["sub_s"] = {1, 2, 3},
|
||||||
|
["sub_u"] = {1, 2, 3},
|
||||||
|
["and_i"] = {1, 2},
|
||||||
|
["and"] = {1, 2, 3},
|
||||||
|
["or_i"] = {1, 2},
|
||||||
|
["or_i_self"] = {1},
|
||||||
|
["or"] = {1, 2, 3},
|
||||||
|
["or_self"] = {1, 2},
|
||||||
|
["xor_i"] = {1, 2},
|
||||||
|
["xor"] = {1, 2, 3},
|
||||||
|
["slt_s"] = {1, 2, 3},
|
||||||
|
["slt_u"] = {1, 2, 3},
|
||||||
|
["slt_si"] = {1, 2},
|
||||||
|
["slt_ui"] = {1, 2},
|
||||||
|
["mult_s"] = {1, 2},
|
||||||
|
["mult_u"] = {1, 2},
|
||||||
|
["div_s"] = {1, 2},
|
||||||
|
["div_u"] = {1, 2},
|
||||||
|
-- Shifts: shift_lleft(rd, rt, shamt); the rt operand is the value, rd is dest.
|
||||||
|
["shift_lleft"] = {1, 2},
|
||||||
|
["shift_lright"] = {1, 2},
|
||||||
|
["shift_aright"] = {1, 2},
|
||||||
|
["shift_lleft_self"] = {1},
|
||||||
|
-- Loads: load_word(rt, base, off); the rt operand is the destination (it's written, not read) and base + off are non-GPR operands.
|
||||||
|
-- The check treats the rt operand as a write, so the read-positions table for `load_*` is empty.
|
||||||
|
["load_word"] = {},
|
||||||
|
["load_half_u"] = {},
|
||||||
|
["load_byte_u"] = {},
|
||||||
|
["load_half"] = {},
|
||||||
|
["load_byte"] = {},
|
||||||
|
["load_upper_i"] = {},
|
||||||
|
["load_ui"] = {},
|
||||||
|
-- Stores write to memory; base + rt operands are non-read for load-delay purposes.
|
||||||
|
["store_word"] = {},
|
||||||
|
["store_half"] = {},
|
||||||
|
["store_byte"] = {},
|
||||||
|
-- Branches read rs (+ rt for beq/bne). The branch delay slot is out of scope.
|
||||||
|
["branch_equal"] = {1, 2},
|
||||||
|
["branch_ne"] = {1, 2},
|
||||||
|
["branch_le_zero"] = {1},
|
||||||
|
["branch_lt_zero"] = {1},
|
||||||
|
["branch_ge_zero"] = {1},
|
||||||
|
["branch_gt_zero"] = {1},
|
||||||
|
-- Jumps / link: jr / jalr read rs only (the target). RD is the destination link.
|
||||||
|
["jump_reg"] = {1},
|
||||||
|
["jump_link"] = {1},
|
||||||
|
["call_reg"] = {1},
|
||||||
|
["call_addr"] = {},
|
||||||
|
["jump"] = {},
|
||||||
|
-- mask_upper is a 2-word macro: shift_lleft then shift_lright. The first reads rt.
|
||||||
|
["mask_upper"] = {1, 2},
|
||||||
|
-- move from/to HI/LO.
|
||||||
|
["mov_from_high"] = {},
|
||||||
|
["mov_from_low"] = {},
|
||||||
|
["mov_to_high"] = {1},
|
||||||
|
["mov_to_low"] = {1},
|
||||||
|
-- GTE transfers / loads / stores / commands: the relevant table values live in the check itself.
|
||||||
|
-- `gte_mv_to_*` writes its rt operand; `gte_mv_from_*` writes its rt operand; `gte_*` commands are atomic-from-the-CPU-POV
|
||||||
|
-- once they issue (the CPU holds until the command completes, so load-delay violations don't surface here).
|
||||||
|
["gte_mv_from_data_r"] = {},
|
||||||
|
["gte_mv_from_ctrl_r"] = {},
|
||||||
|
["gte_mv_to_data_r"] = {},
|
||||||
|
["gte_mv_to_ctrl_r"] = {},
|
||||||
|
["gte_lw"] = {},
|
||||||
|
["gte_sw"] = {},
|
||||||
|
["shift_lleft_var"] = {1, 2, 3}, -- rd, rt, rs (variable shift amount)
|
||||||
|
["shift_aright_var"] = {1, 2, 3},
|
||||||
|
}
|
||||||
|
|
||||||
|
-- GP0 packet sizes (total words including the 1-word tag) per GP0 cmd byte.
|
||||||
|
-- Per PSX-SPX `docs/psx-spx/docs/graphicsprocessingunitgpu.md` §"GPU Render Polygon Commands":
|
||||||
|
-- Each polygon command's word count = 1 (tag/cmd) + per-vertex (vertex + optional color + optional UV).
|
||||||
|
-- F3: cmd + 3 vertices = 4 words; +1 tag = 5
|
||||||
|
-- F4: cmd + 4 vertices = 5 words; +1 tag = 6
|
||||||
|
-- G3: cmd + 3×(color + vertex) = 6 words; +1 tag = 7
|
||||||
|
-- G4: cmd + 4×(color + vertex) = 8 words; +1 tag = 9
|
||||||
|
-- FT3: cmd + tpage + clut + 3×(vertex + UV) = 7 words; +1 tag = 8
|
||||||
|
-- FT4: cmd + tpage + clut + 4×(vertex + UV) = 9 words; +1 tag = 10
|
||||||
|
-- GT3: cmd + tpage + clut + 3×(color + vertex + UV) = 9 words; +1 tag = 10
|
||||||
|
-- GT4: cmd + tpage + clut + 4×(color + vertex + UV) = 12 words; +1 tag = 13
|
||||||
|
--
|
||||||
|
-- Cross-checked against code/duffle/gp.h struct sizes + the set_poly_* macros
|
||||||
|
-- (which encode "len" = "words after tag"):
|
||||||
|
-- set_poly_f3(p) -> set_len(p, 4) -> 5 total GP0 0x20
|
||||||
|
-- set_poly_ft3(p) -> set_len(p, 7) -> 8 total GP0 0x24
|
||||||
|
-- set_poly_f4(p) -> set_len(p, 5) -> 6 total GP0 0x28
|
||||||
|
-- set_poly_ft4(p) -> set_len(p, 9) -> 10 total GP0 0x2C
|
||||||
|
-- set_poly_g3(p) -> set_len(p, 6) -> 7 total GP0 0x30
|
||||||
|
-- set_poly_gt3(p) -> set_len(p, 9) -> 10 total GP0 0x34
|
||||||
|
-- set_poly_g4(p) -> set_len(p, 8) -> 9 total GP0 0x38
|
||||||
|
-- set_poly_gt4(p) -> set_len(p, 12) -> 13 total GP0 0x3C
|
||||||
|
M.GP0_CMD_SIZE = {
|
||||||
|
[0x20] = 5, -- Poly_F3
|
||||||
|
[0x24] = 8, -- Poly_FT3
|
||||||
|
[0x28] = 6, -- Poly_F4
|
||||||
|
[0x2C] = 10, -- Poly_FT4
|
||||||
|
[0x30] = 7, -- Poly_G3
|
||||||
|
[0x34] = 10, -- Poly_GT3
|
||||||
|
[0x38] = 9, -- Poly_G4
|
||||||
|
[0x3C] = 13, -- Poly_GT4
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Shape suffix (after `ac_format_` / `mac_format_` prefix) -> GP0 cmd byte.
|
||||||
|
-- Lets the static-analysis check derive the cmd byte from a macro name like `mac_format_g4_color` -> `g4` -> 0x38 -> 9 expected words.
|
||||||
|
M.GP0_CMD_BY_SHAPE = {
|
||||||
|
["f3"] = 0x20, ["ft3"] = 0x24,
|
||||||
|
["f4"] = 0x28, ["ft4"] = 0x2C,
|
||||||
|
["g3"] = 0x30, ["gt3"] = 0x34,
|
||||||
|
["g4"] = 0x38, ["gt4"] = 0x3C,
|
||||||
|
}
|
||||||
|
|
||||||
|
M.UNKNOWN_INSTRUCTION_CYCLES = 1
|
||||||
|
|
||||||
|
-- Hardware-relation policy table.
|
||||||
|
--
|
||||||
|
-- The forward walker in `passes/static_analysis.lua::analyze_hardware_relations` reads every emitted word_event, matches its `encoder` against `row.token`, and:
|
||||||
|
-- * stages the event as a producer in `atom.paths.forward_state`; or
|
||||||
|
-- * matches it as a consumer against pending producers and records a hazard on `atom.paths.hazards` when the gap is below `visibility.required`.
|
||||||
|
--
|
||||||
|
-- Each row is the contract for one CPU-to-coprocessor transfer semantic (the coprocessor-to-CPU path mirrors the same shape).
|
||||||
|
-- The `reads` / `writes` sub-tables carry the argument positions the analyzer inspects:
|
||||||
|
-- * `writes.arg` is the destination operand (the producer's effect); the analyzer stages this register as a pending producer.
|
||||||
|
-- * `reads` (when present) lists the operand positions the same token reads back from hardware; for MTC2 / CTC2 the producer reads the GPR source it is loading from.
|
||||||
|
-- The `fanout_to` field (MTC2-IRGB row only) tells the consumer-match logic which downstream COP2 registers are transitively updated by the write.
|
||||||
|
--
|
||||||
|
-- Visibility semantics:
|
||||||
|
-- * `kind = "post_producer_words"` means the consumer observes the producer's effect after `required` independent emitted words that are
|
||||||
|
-- strictly between the producer and the consumer. The producer's own emitted slot is implicit (it counts as the slot of issue, not toward `required`)
|
||||||
|
-- per the PSX-SPX rule: "Store delays are counted in numbers of clock cycles (not in numbers of opcodes).
|
||||||
|
-- For 3 cycle delay, one must usually insert 3 cached opcodes (or one uncached opcode)."
|
||||||
|
-- * `required` is the minimum count of intervening emitted words between producer and consumer.
|
||||||
|
-- `required = 0` permits the consumer on the very next slot; `required < 0` would place the consumer on the same slot as the producer
|
||||||
|
-- and is reserved for future "self-retires" relations.
|
||||||
|
--
|
||||||
|
-- Evidence:
|
||||||
|
-- * `evidence.confidence` is one of `"exact"`, `"conservative"`, `"unknown"`. The severity comes from `violation_kind`;
|
||||||
|
-- A hardware measurement that the vendor caveats may still classify as `"conservative"` even when the underlying timing is numerically known.
|
||||||
|
-- * `evidence.source` is the upstream reference (file + line range) the row is sourced from. New rows must carry this citation.
|
||||||
|
--
|
||||||
|
-- Consumers:
|
||||||
|
-- * passes/static_analysis.lua::analyze_hardware_relations (forward walker).
|
||||||
|
-- * passes/static_analysis.lua::transfer_hazards CHECK_RULES reader (renders hazards onto `findings`).
|
||||||
|
-- This table is consumed by the hardware-relation analyzer and hazard renderer.
|
||||||
|
M.HARDWARE_RELATIONS = {
|
||||||
|
-- CPU → COP2 data register (MTC2). The ordinary default is 2 cached words between producer and consumer (cpuspecifications.md:407-419).
|
||||||
|
{
|
||||||
|
id = "mtc2_gpr_visibility",
|
||||||
|
semantic = "MTC2",
|
||||||
|
token = "gte_mv_to_data_r",
|
||||||
|
direction = "gpr_to_cop2_data",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop2.data", arg = 2 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 2 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- CPU → COP2 data register when the destination is C2_IRGB (data 28).
|
||||||
|
-- C2_IRGB drives the IR1/IR2/IR3 color-conversion fan-out, which extends the propagation delay to 3 cached words.
|
||||||
|
-- `destination_match = "C2_IRGB"` is the row's filter; the analyzer consults this when the producer's destination operand equals "C2_IRGB".
|
||||||
|
-- C2_ORGB (data 29) is read-only and is never classified as a writable fan-out destination.
|
||||||
|
{
|
||||||
|
id = "mtc2_irgb_visibility",
|
||||||
|
semantic = "MTC2",
|
||||||
|
token = "gte_mv_to_data_r",
|
||||||
|
direction = "gpr_to_cop2_data",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop2.data", arg = 2 },
|
||||||
|
destination_match = "C2_IRGB",
|
||||||
|
fanout_to = { "C2_IR1", "C2_IR2", "C2_IR3" },
|
||||||
|
visibility = { kind = "post_producer_words", required = 3 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- CPU → COP2 control register (CTC2). Ordinary minimum 2;
|
||||||
|
-- no IRGB-style fan-out exists for control registers (per spec §3.6: only C2_IRGB has the 3-cycle fan-out on the data side).
|
||||||
|
{
|
||||||
|
id = "ctc2_gpr_visibility",
|
||||||
|
semantic = "CTC2",
|
||||||
|
token = "gte_mv_to_ctrl_r",
|
||||||
|
direction = "gpr_to_cop2_control",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop2.ctrl", arg = 2 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 2 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- COP2 data → GPR (MFC2). One cached slot between the transfer and the first GPR consumer;
|
||||||
|
-- the GPR is not updated until the instruction AFTER the MFC2 completes (geometrytransformationenginegte.md:29-32).
|
||||||
|
{
|
||||||
|
id = "mfc2_gpr_visibility",
|
||||||
|
semantic = "MFC2",
|
||||||
|
token = "gte_mv_from_data_r",
|
||||||
|
direction = "cop2_data_to_gpr",
|
||||||
|
reads = { domain = "cop2.data", arg = 2 },
|
||||||
|
writes = { domain = "gpr", arg = 1 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 1 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "geometrytransformationenginegte.md:29-32",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- COP2 control → GPR (CFC2). Same delay as MFC2 (cpuspecifications.md treats the two load-from-COP2 paths symmetrically).
|
||||||
|
{
|
||||||
|
id = "cfc2_gpr_visibility",
|
||||||
|
semantic = "CFC2",
|
||||||
|
token = "gte_mv_from_ctrl_r",
|
||||||
|
direction = "cop2_control_to_gpr",
|
||||||
|
reads = { domain = "cop2.ctrl", arg = 2 },
|
||||||
|
writes = { domain = "gpr", arg = 1 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 1 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:382-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- COP0 control → GPR (MFC0).
|
||||||
|
-- One cached slot; the analyzer treats `sys_mov_from_cop0(rt, 12)` (the SR/CU2 transfer) as the same shape as the COP2 load-delay path.
|
||||||
|
-- The semantic-level SR/CU2 transition models the load delay;
|
||||||
|
-- SR.CU2 bounded-value propagation is modeled separately).
|
||||||
|
{
|
||||||
|
id = "mfc0_gpr_visibility",
|
||||||
|
semantic = "MFC0",
|
||||||
|
token = "sys_mov_from_cop0",
|
||||||
|
direction = "cop0_control_to_gpr",
|
||||||
|
reads = { domain = "cop0.ctrl", arg = 2 },
|
||||||
|
writes = { domain = "gpr", arg = 1 },
|
||||||
|
visibility = { kind = "post_producer_words", required = 1 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:171-178",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
},
|
||||||
|
-- Memory -> COP2 data register (LWC2).
|
||||||
|
-- The memory-side timing is not measured by the vendored GTE latch experiment, so this relation has no numeric retirement threshold.
|
||||||
|
-- The LWC2 destination has TWO retirement regimes (per PSX-SPX):
|
||||||
|
-- * GTE-command consumer (`gte_cmdw_*`): the GTE pipeline LATCHES the LWC2 result, so a `gte_cmdw_*`
|
||||||
|
-- in the very next slot uses the latched value. Gap = 0 is allowed. (Per `docs/psx-spx/docs/gtepipelinetimings.md:271-274`.)
|
||||||
|
-- * Any other consumer: standard MIPS load delay applies. Gap = 1 required. (Per `docs/psx-spx/docs/cpuspecifications.md:407-419`.)
|
||||||
|
-- Two separate relations so the walker can dispatch by consumer type and emit different severities
|
||||||
|
-- (the GTE-command path is `info` because the latch is intentional; the non-GTE-consumer path is `error` because the missing nop is a real bug).
|
||||||
|
{
|
||||||
|
id = "lwc2_to_gte_command",
|
||||||
|
semantic = "LWC2_to_GTE",
|
||||||
|
token = "gte_lw",
|
||||||
|
direction = "memory_to_cop2_data",
|
||||||
|
reads = { domain = "memory", arg = 2 },
|
||||||
|
writes = { domain = "cop2.data", arg = 1 },
|
||||||
|
required = 0, -- GTE-command consumer: gap = 0 OK (latched).
|
||||||
|
evidence = {
|
||||||
|
confidence = "measured",
|
||||||
|
source = "gtepipelinetimings.md:271-274",
|
||||||
|
},
|
||||||
|
violation_kind = "info",
|
||||||
|
clear_on_consumer = true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
id = "lwc2_to_other_consumer",
|
||||||
|
semantic = "LWC2_to_other",
|
||||||
|
token = "gte_lw",
|
||||||
|
direction = "memory_to_cop2_data",
|
||||||
|
reads = { domain = "memory", arg = 2 },
|
||||||
|
writes = { domain = "cop2.data", arg = 1 },
|
||||||
|
required = 1, -- Non-GTE-consumer: standard MIPS load delay.
|
||||||
|
evidence = {
|
||||||
|
confidence = "inferred",
|
||||||
|
source = "cpuspecifications.md:407-419",
|
||||||
|
},
|
||||||
|
violation_kind = "error",
|
||||||
|
clear_on_consumer = true,
|
||||||
|
},
|
||||||
|
-- COP2 data register -> memory (SWC2). A read of C2 state, not a CPU-to-COP2 write.
|
||||||
|
-- The policy row stays in for direction/provenance; staging it as a later command-input producer is suppressed.
|
||||||
|
{
|
||||||
|
id = "swc2_memory_write",
|
||||||
|
semantic = "SWC2",
|
||||||
|
token = "gte_sw",
|
||||||
|
direction = "cop2_data_to_memory",
|
||||||
|
reads = { domain = "cop2.data", arg = 1 },
|
||||||
|
writes = { domain = "memory", arg = 2 },
|
||||||
|
visibility = { kind = "none", required = 0 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "exact",
|
||||||
|
source = "cpuspecifications.md:79",
|
||||||
|
},
|
||||||
|
violation_kind = "info",
|
||||||
|
stage = false,
|
||||||
|
},
|
||||||
|
-- MTC0 Status/SR.CU2. The ordinary COP0 store has no general store-delay relation;
|
||||||
|
-- this row feeds the dedicated CU2 transition logic in the same forward walk and is therefore not staged in `pending`.
|
||||||
|
{
|
||||||
|
id = "mtc0_cu2_visibility",
|
||||||
|
semantic = "MTC0",
|
||||||
|
token = "sys_mov_to_cop0",
|
||||||
|
direction = "gpr_to_cop0_status",
|
||||||
|
reads = { domain = "gpr", arg = 1 },
|
||||||
|
writes = { domain = "cop0.status", arg = 2 },
|
||||||
|
status_register = 12,
|
||||||
|
visibility = { kind = "post_producer_words", required = 2 },
|
||||||
|
evidence = {
|
||||||
|
confidence = "conservative",
|
||||||
|
source = "cpuspecifications.md:543,625-628",
|
||||||
|
},
|
||||||
|
violation_kind = "warning",
|
||||||
|
stage = false,
|
||||||
|
cu2_transition = true,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Bounded Status/SR.CU2 transition policy.
|
||||||
|
-- The value lattice and the transition consumer both read this immutable row; no second value pass is permitted.
|
||||||
|
-- The source says the enable/disable transition takes "2 clock cycles or so", so the boundary is conservative rather than exact.
|
||||||
|
M.CU2_TRANSITION_POLICY = {
|
||||||
|
status_register = 12,
|
||||||
|
enable_bit = 0x40000000,
|
||||||
|
required = 2,
|
||||||
|
visibility_kind = "post_producer_words",
|
||||||
|
evidence = {
|
||||||
|
confidence = "conservative",
|
||||||
|
source = "cpuspecifications.md:543,625-628",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
return M
|
||||||
@@ -22,12 +22,10 @@ local M = {}
|
|||||||
local CACHE_KEY = "__duffle_repo_root__"
|
local CACHE_KEY = "__duffle_repo_root__"
|
||||||
|
|
||||||
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
--- Resolve the repo root from this script's own path. Zero shell spawn.
|
||||||
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the
|
--- `duffle_paths.lua` always lives at `<repo>/scripts/duffle_paths.lua`, so the repo root is the parent of the directory containing this script.
|
||||||
--- parent of the directory containing this script. We derive it directly from `debug.getinfo(1, "S").source`
|
--- We derive it directly from `debug.getinfo(1, "S").source` (returns `@<path>` for the currently-running chunk).
|
||||||
--- (returns `@<path>` for the currently-running chunk).
|
|
||||||
---
|
---
|
||||||
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source),
|
--- If `debug.getinfo` can't parse this script's path (shouldn't happen — dofile always populates source), return nil and let `M.setup()` fail loud.
|
||||||
--- return nil and let `M.setup()` fail loud.
|
|
||||||
--- @return string|nil
|
--- @return string|nil
|
||||||
local function find_repo_root()
|
local function find_repo_root()
|
||||||
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
if package.loaded[CACHE_KEY] then return package.loaded[CACHE_KEY] end
|
||||||
@@ -51,17 +49,13 @@ end
|
|||||||
---
|
---
|
||||||
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
||||||
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
||||||
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
|
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`, which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
||||||
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
|
||||||
function M.setup()
|
function M.setup()
|
||||||
local repo_root = find_repo_root()
|
local repo_root = find_repo_root()
|
||||||
if not repo_root then
|
if not repo_root then
|
||||||
-- Unreachable in practice: find_repo_root() derives the repo root from this script's
|
-- Unreachable in practice: find_repo_root() derives the repo root from this script's own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
||||||
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
-- A nil return means the source path did not match the expected <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo" condition.
|
||||||
-- A nil return means the source path did not match the expected
|
-- os.exit(2) is retained so a real failure surfaces loud rather than silently producing an unconfigured module table.
|
||||||
-- <repo>/scripts/duffle_paths.lua layout — a packaging bug, not a "missing git repo"
|
|
||||||
-- condition. os.exit(2) is retained so a real failure surfaces loud rather than
|
|
||||||
-- silently producing an unconfigured module table.
|
|
||||||
os.exit(2)
|
os.exit(2)
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -86,6 +80,6 @@ end
|
|||||||
-- Run the setup as a side effect.
|
-- Run the setup as a side effect.
|
||||||
M.setup()
|
M.setup()
|
||||||
|
|
||||||
-- Now that package.path includes scripts/, `require("duffle")` resolves. Return the duffle module
|
-- Now that package.path includes scripts/, `require("duffle")` resolves.
|
||||||
-- so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
-- Return the duffle module so callers can do `local duffle = dofile(...duffle_paths.lua)` in one line.
|
||||||
return require("duffle")
|
return require("duffle")
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -468,8 +468,7 @@ function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
|||||||
-- 20: <children>
|
-- 20: <children>
|
||||||
if body_end - body_start >= 20 then
|
if body_end - body_start >= 20 then
|
||||||
-- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte);
|
-- read_ref_sig8 / write_u32_le / etc. are 1-indexed (string:byte);
|
||||||
-- pos / body_start / body_end are 0-based wire offsets, so the
|
-- pos / body_start / body_end are 0-based wire offsets, so the 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
|
||||||
-- 1-indexed byte at 0-based wire offset X is string:byte(X + 1).
|
|
||||||
-- Per DWARF5 §7.5.6, the type_unit body is laid out as:
|
-- Per DWARF5 §7.5.6, the type_unit body is laid out as:
|
||||||
-- byte 0-1: version (2)
|
-- byte 0-1: version (2)
|
||||||
-- byte 2: unit_type (1) -- DW_UT_type = 0x02
|
-- byte 2: unit_type (1) -- DW_UT_type = 0x02
|
||||||
@@ -612,13 +611,13 @@ function M.read_elf_sections(elf_path, section_names)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
|
--- Read ELF symbol addresses by walking the `.symtab` + `.strtab` sections directly (no `nm` subprocess).
|
||||||
--- Returns a map `{name -> {addr, size_bytes}}` for every `code_<name>` symbol.
|
--- Returns a map `{name -> {addr, size_bytes}}` for every defined symbol.
|
||||||
---
|
---
|
||||||
--- **Conventions:**
|
--- **Conventions:**
|
||||||
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
|
--- - ELF32 symtab entry = 16 bytes (`st_name:4 + st_value:4 + st_size:4 + st_info:1 + st_other:1 + st_shndx:2`); offsets within each entry are zero-based wire offsets.
|
||||||
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
|
--- - Direct Lua `string.byte`/`string.sub`/`string.find` boundaries receive `+ 1`.
|
||||||
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
||||||
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
--- - Keys are the ELF symbol names as written (the C ident).
|
||||||
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
||||||
---
|
---
|
||||||
--- @param elf_path Path
|
--- @param elf_path Path
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ define tape_atoms
|
|||||||
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
echo "[gdb_tape_atoms] STUB: run .\\build_psyq.ps1 to regenerate, then re-source this file."
|
||||||
end
|
end
|
||||||
document tape_atoms
|
document tape_atoms
|
||||||
List every tape atom symbol in the loaded ELF (code_<name>) with its .rodata address and word count.
|
List every tape atom symbol in the loaded ELF with its .rodata address and word count.
|
||||||
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
STUB state: runtime file not sourced. Run build_psyq.ps1 to regenerate.
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
@@ -1,10 +1,9 @@
|
|||||||
# scripts/launch_pcsx_debug.ps1
|
# scripts/launch_pcsx_debug.ps1
|
||||||
#
|
#
|
||||||
# One-shot launcher for debug sessions: starts pcsx-redux with the .ps-exe
|
# One-shot launcher for debug sessions:
|
||||||
# loaded, the gdb stub enabled, AND the pcsx_debug_helper Lua plugin loaded
|
# Starts pcsx-redux with the .ps-exe loaded, the gdb stub enabled,
|
||||||
# so external CLI tools (gdb's `shell` command, etc.)
|
# AND the pcsx_debug_helper Lua plugin loaded so external CLI tools (gdb's `shell` command, etc.)
|
||||||
# can read GTE state via http://localhost:8080/api/v1/lua/gte
|
# can read GTE state via http://localhost:8080/api/v1/lua/gte (the gdb stub doesn't expose COP2 at all).
|
||||||
# (the gdb stub doesn't expose COP2 at all).
|
|
||||||
#
|
#
|
||||||
# usage:
|
# usage:
|
||||||
# .\scripts\launch_pcsx_debug.ps1
|
# .\scripts\launch_pcsx_debug.ps1
|
||||||
@@ -84,7 +83,8 @@ try {
|
|||||||
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
$r = Invoke-WebRequest -Uri "http://localhost:$WebPort/api/v1/lua/gte" -UseBasicParsing -TimeoutSec 5
|
||||||
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
$firstLine = ([System.Text.Encoding]::UTF8.GetString($r.Content) -split "`n")[0]
|
||||||
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
Write-Host "GTE handler OK: $firstLine" -ForegroundColor Green
|
||||||
} catch {
|
}
|
||||||
|
catch {
|
||||||
Write-Warning "GTE handler NOT responding: $_"
|
Write-Warning "GTE handler NOT responding: $_"
|
||||||
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
Write-Host "Check the pcsx-redux Lua Console for debug cli messages." -ForegroundColor Yellow
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -115,7 +115,7 @@ end
|
|||||||
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
|
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
|
||||||
--- @param pipe_ctx PipeCtx
|
--- @param pipe_ctx PipeCtx
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_unique_annotation(pipe_ctx, findings)
|
local function check_unique_annotation(_item, pipe_ctx, findings)
|
||||||
for name, n in pairs(pipe_ctx.annot_counts) do
|
for name, n in pairs(pipe_ctx.annot_counts) do
|
||||||
if n > 1 then
|
if n > 1 then
|
||||||
findings.errors[#findings.errors + 1] = {
|
findings.errors[#findings.errors + 1] = {
|
||||||
@@ -147,7 +147,8 @@ end
|
|||||||
--- @param m MacroEntry
|
--- @param m MacroEntry
|
||||||
--- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
|
--- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_macro_word_drift(m, wc, findings)
|
local function check_macro_word_drift(m, pipe_ctx, findings)
|
||||||
|
local wc = (pipe_ctx and pipe_ctx.word_counts) or {}
|
||||||
local declared = wc[m.name]
|
local declared = wc[m.name]
|
||||||
if not declared then
|
if not declared then
|
||||||
findings.errors[#findings.errors + 1] = {
|
findings.errors[#findings.errors + 1] = {
|
||||||
@@ -429,40 +430,17 @@ local CHECK_RULES = {
|
|||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @return PipeCtx
|
--- @return PipeCtx
|
||||||
local function build_corpus_pipe_ctx(ctx)
|
local function build_corpus_pipe_ctx(ctx)
|
||||||
local corpus = ctx.shared and ctx.shared.corpus
|
local view = duffle.corpus_view(ctx)
|
||||||
if not corpus then
|
|
||||||
error("annotation requires ctx.shared.corpus "
|
|
||||||
.. "(the canonical corpus is the source of truth; "
|
|
||||||
.. "no per-source fallback is supported)", 0)
|
|
||||||
end
|
|
||||||
|
|
||||||
-- `corpus.atom_infos` preserves source order and duplicates; I precompute counts here for `check_unique_annotation` and the per-source checks.
|
|
||||||
local annot_counts = {}
|
local annot_counts = {}
|
||||||
for _, info in ipairs(corpus.atom_infos or {}) do
|
for _, info in ipairs(view.atom_infos) do
|
||||||
if info and info.atom_name then
|
if info and info.atom_name then
|
||||||
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
|
annot_counts[info.atom_name] = (annot_counts[info.atom_name] or 0) + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
view.annot_counts = annot_counts
|
||||||
-- Every consumer of these fields observes mutations via the canonical corpus without independently mutable registry construction.
|
view.atom_infos_list = view.atom_infos
|
||||||
return {
|
view.word_counts = ctx.shared.corpus.word_counts or {}
|
||||||
-- Cross-source lookup tables from corpus.
|
return view
|
||||||
register_alias_registry = corpus.register_alias_registry or {},
|
|
||||||
type_name_registry = corpus.type_name_registry or {},
|
|
||||||
atom_views = corpus.atom_views or {},
|
|
||||||
atom_ctxs = corpus.atom_ctxs or {},
|
|
||||||
atom_phases = corpus.atom_phases or {},
|
|
||||||
binds_by_name = corpus.binds_by_name or {},
|
|
||||||
atoms_by_name = corpus.atoms_by_name or {},
|
|
||||||
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
|
||||||
atom_infos_list = corpus.atom_infos or {},
|
|
||||||
-- Corpus-wide annotation count aggregation (post-rule consumes this).
|
|
||||||
annot_counts = annot_counts,
|
|
||||||
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
|
||||||
collisions = corpus.collisions or {},
|
|
||||||
-- `check_macro_word_drift` reads `corpus.word_counts`, populated by word_count_eval.run.
|
|
||||||
word_counts = corpus.word_counts or {},
|
|
||||||
}
|
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
||||||
@@ -477,8 +455,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
|||||||
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
-- Project the pre-scanned atoms to the AtomEntry shape this pass needs.
|
||||||
local atoms = {}
|
local atoms = {}
|
||||||
for _, a in ipairs(scan.atoms) do
|
for _, a in ipairs(scan.atoms) do
|
||||||
if a.kind == "atom" then
|
if a.kind == "atom" or a.kind == "atom_proc" then
|
||||||
atoms[#atoms + 1] = { line = a.line, name = a.raw_name }
|
atoms[#atoms + 1] = { line = a.line, name = a.raw_name or a.name }
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -536,38 +514,28 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
|||||||
|
|
||||||
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
|
-- THE per-annotation pipeline. ONE loop. CHECK_RULES dispatches per_annot rules.
|
||||||
for _, a in ipairs(annots) do
|
for _, a in ipairs(annots) do
|
||||||
for _, rule in ipairs(CHECK_RULES) do
|
duffle.run_check_rules(CHECK_RULES, "per_annot", a, pipe_ctx, findings)
|
||||||
if rule.per_annot then rule.per_annot(a, pipe_ctx, findings) end
|
|
||||||
end
|
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
|
-- Post-loop rules (one-shot checks that need full-corpus aggregation in pipe_ctx).
|
||||||
for _, rule in ipairs(CHECK_RULES) do
|
duffle.run_check_rules(CHECK_RULES, "post", nil, pipe_ctx, findings)
|
||||||
if rule.post then rule.post(pipe_ctx, findings) end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
|
-- scan_source records each marker in scan.debug_skip_markers; this loop validates each record independently and emits at most one error per marker.
|
||||||
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
|
-- Valid markers stamp `debug_skip = true` on the following atom or component declaration, which downstream consumers read directly.
|
||||||
local skip_markers = scan.debug_skip_markers or {}
|
local skip_markers = scan.debug_skip_markers or {}
|
||||||
for _, marker in ipairs(skip_markers) do
|
for _, marker in ipairs(skip_markers) do
|
||||||
for _, rule in ipairs(CHECK_RULES) do
|
duffle.run_check_rules(CHECK_RULES, "per_skip_marker", marker, pipe_ctx, findings)
|
||||||
if rule.per_skip_marker then rule.per_skip_marker(marker, pipe_ctx, findings) end
|
|
||||||
end
|
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
|
-- Per-macro rules (TAPE_WORDS vs WORD_COUNT drift).
|
||||||
local wc = corpus_pipe_ctx.word_counts
|
pipe_ctx.word_counts = corpus_pipe_ctx.word_counts
|
||||||
for _, m in ipairs(scan.macros) do
|
for _, m in ipairs(scan.macros) do
|
||||||
for _, rule in ipairs(CHECK_RULES) do
|
duffle.run_check_rules(CHECK_RULES, "per_macro", m, pipe_ctx, findings)
|
||||||
if rule.per_macro then rule.per_macro(m, wc, findings) end
|
|
||||||
end
|
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
|
-- Per-source rules (reg defaults, atom_view layout, compute-register type overrides, Binds_* field uniqueness).
|
||||||
-- Each per_source rule sees the full scan payload via pipe_ctx.
|
-- Each per_source rule sees the full scan payload via pipe_ctx.
|
||||||
for _, rule in ipairs(CHECK_RULES) do
|
duffle.run_check_rules(CHECK_RULES, "per_source", src, pipe_ctx, findings)
|
||||||
if rule.per_source then rule.per_source(src, pipe_ctx, findings) end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Information summary (always emitted).
|
-- Information summary (always emitted).
|
||||||
findings.info[#findings.info + 1] = {
|
findings.info[#findings.info + 1] = {
|
||||||
|
|||||||
@@ -83,6 +83,7 @@ local function canonical_word_entries(atom)
|
|||||||
line = event.call_line or item.line or 0,
|
line = event.call_line or item.line or 0,
|
||||||
text = event.call_text or item.call_text or "",
|
text = event.call_text or item.call_text or "",
|
||||||
body_line = event.body_line or item.body_line or item.line or 0,
|
body_line = event.body_line or item.body_line or item.line or 0,
|
||||||
|
gpr_keys = event.gpr_keys,
|
||||||
invocation = (event.outermost_invocation_id
|
invocation = (event.outermost_invocation_id
|
||||||
and paths.invocations
|
and paths.invocations
|
||||||
and paths.invocations[event.outermost_invocation_id]) or nil,
|
and paths.invocations[event.outermost_invocation_id]) or nil,
|
||||||
@@ -260,12 +261,12 @@ local function append_gdb_commands(lines, matched)
|
|||||||
for _, a in ipairs(matched) do
|
for _, a in ipairs(matched) do
|
||||||
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
-- gdb 12.1 quirk: literals in printf args require an attached target.
|
||||||
-- Use the per-atom convenience vars set above as printf args.
|
-- Use the per-atom convenience vars set above as printf args.
|
||||||
lines[#lines + 1] = string.format(' printf " code_%%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
lines[#lines + 1] = string.format(' printf " %%-32s @ 0x%%08x %%4d words\\n", $__atom_name_%d, $__atom_addr_%d, $__atom_words_%d',
|
||||||
a.idx, a.idx, a.idx)
|
a.idx, a.idx, a.idx)
|
||||||
end
|
end
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = "document tape_atoms"
|
lines[#lines + 1] = "document tape_atoms"
|
||||||
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF (code_<name>) with .rodata addr + word count."
|
lines[#lines + 1] = " List every tape atom symbol in the loaded ELF with .rodata addr + word count."
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = ""
|
lines[#lines + 1] = ""
|
||||||
|
|
||||||
@@ -284,10 +285,10 @@ local function append_gdb_commands(lines, matched)
|
|||||||
for _, a in ipairs(matched) do
|
for _, a in ipairs(matched) do
|
||||||
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
lines[#lines + 1] = string.format("define break_atom_%s", a.name)
|
||||||
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
lines[#lines + 1] = string.format(" break *$__atom_addr_%d", a.idx)
|
||||||
lines[#lines + 1] = string.format(' printf " Breakpoint set at code_%s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
lines[#lines + 1] = string.format(' printf " Breakpoint set at %s (0x%%08x)\\n", $__atom_addr_%d', a.name, a.idx)
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
lines[#lines + 1] = string.format("document break_atom_%s", a.name)
|
||||||
lines[#lines + 1] = string.format(" Set a breakpoint at code_%s.", a.name)
|
lines[#lines + 1] = string.format(" Set a breakpoint at %s.", a.name)
|
||||||
lines[#lines + 1] = "end"
|
lines[#lines + 1] = "end"
|
||||||
lines[#lines + 1] = ""
|
lines[#lines + 1] = ""
|
||||||
end
|
end
|
||||||
@@ -322,7 +323,7 @@ local function append_gdb_commands(lines, matched)
|
|||||||
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
-- Precompute end_addr (gdb 12.1's expression evaluator chokes on `addr + words*4`).
|
||||||
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
lines[#lines + 1] = string.format(" set $__end_%d = $__atom_addr_%d + $__atom_words_%d * 4", a.idx, a.idx, a.idx)
|
||||||
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
lines[#lines + 1] = string.format(" if $__pc >= $__atom_addr_%d && $__pc < $__end_%d", a.idx, a.idx)
|
||||||
lines[#lines + 1] = string.format(' printf "atom: code_%%s\\n", $__atom_name_%d', a.idx)
|
lines[#lines + 1] = string.format(' printf "atom: %%s\\n", $__atom_name_%d', a.idx)
|
||||||
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
lines[#lines + 1] = ' printf "addr: 0x%08x\\n", $__pc'
|
||||||
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
lines[#lines + 1] = string.format(" set $__word = ($__pc - $__atom_addr_%d) / 4", a.idx)
|
||||||
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
lines[#lines + 1] = string.format(' printf "word: %%d/%%d\\n", $__word, $__atom_words_%d', a.idx)
|
||||||
@@ -491,8 +492,19 @@ function M.render_atom_source_map(atom)
|
|||||||
local lines = {}
|
local lines = {}
|
||||||
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
lines[#lines + 1] = string.format("ATOM %s %d", (atom.raw_name or atom.name), total)
|
||||||
for _, entry in ipairs(entries) do
|
for _, entry in ipairs(entries) do
|
||||||
lines[#lines + 1] = string.format("WORD %d LINE %d TEXT %s",
|
local word_line = string.format("WORD %d LINE %d TEXT %s",
|
||||||
entry.pos, entry.line, entry.text)
|
entry.pos, entry.line, entry.text)
|
||||||
|
local keys = {}
|
||||||
|
for pos = 1, 16 do
|
||||||
|
local k = entry.gpr_keys and entry.gpr_keys[pos]
|
||||||
|
if type(k) == "string" and k:sub(1, 7) == "reguse:" then
|
||||||
|
keys[#keys + 1] = k
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #keys > 0 then
|
||||||
|
word_line = word_line .. " KEYS " .. table.concat(keys, ",")
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = word_line
|
||||||
end
|
end
|
||||||
lines[#lines + 1] = "ENDATOM"
|
lines[#lines + 1] = "ENDATOM"
|
||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
@@ -529,7 +541,7 @@ function M.render_atom_provenance(atom, wc, rel_path)
|
|||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Pass entry. For each source that declares at least one `MipsAtom_(name)` / `MipsCode code_<name>`,
|
--- Pass entry. For each source that declares at least one tape atom,
|
||||||
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
--- emit two files in `<out_root>/`: `<basename>.atoms.sourcemap.txt` (per-word call-site map) and `<basename>.atoms.provenance.txt`
|
||||||
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
--- (per-word definition + body line, resolved via the outermost `mac_X(...)` invocation).
|
||||||
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
--- When `ctx.flags.gdb_runtime` is true and `ctx.flags.elf_path` exists, also emit the post-link gdb script `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`.
|
||||||
|
|||||||
+28
-40
@@ -27,50 +27,34 @@
|
|||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
|
||||||
--- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
--- THE GPR ALLOCATION POOL — what is allocatable, and (more importantly) WHY
|
-- THE GPR ALLOCATION POOL — what's allocatable, and (more importantly) WHY
|
||||||
--- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
---
|
--
|
||||||
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
|
-- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
|
||||||
--- It allocates from a FIXED 10-register pool.
|
-- The 24-register pool covers R2-R25 (the user/atom allocatable surface):
|
||||||
--- This comment block makes the inclusion AND exclusion criteria obvious so a reader doesn't have
|
-- R_T0..R_T7, R_V0..R_V1, R_A0..A3, R_S0..S7, R_T8..T9.
|
||||||
--- to grep lottes_tape.h + mips.h to understand the design.
|
-- Excluded (and never added to the pool):
|
||||||
---
|
-- R_0 (code 0) — hardwired zero. Cannot be written.
|
||||||
--- ── WHAT'S IN THE POOL (10 GPRs, all caller-trash per the O32 ABI) ────────
|
-- R_AT (code 1) — assembler temporary. Reserved by the MIPS O32 ABI.
|
||||||
--- R_T0..R_T7 (GPR codes 8..15), R_V0..R_V1 (GPR codes 2..3)
|
-- R_A0..A3 — explicitly omitted above even though their integer codes
|
||||||
--- The workhorse of every atom body. The uesr should be aware of atom allocation across atoms they chain.
|
-- map to POOL entries; the pool-construction loop below
|
||||||
--- If they have a collision it means either they didn't saturate the register file optimally for a phase,
|
-- only references the POOL string literals, never the
|
||||||
--- or the may have made the workload to large for the run.
|
-- integer codes, so they are NOT auto-allocated by default.
|
||||||
---
|
-- (A0-A3 become available when the user adds them to
|
||||||
--- ── WHAT'S NOT IN THE POOL — and WHY (the "obvious exclusions") ────────────
|
-- POOL or hardcodes an R_A0 reference in the atom body.)
|
||||||
--- R_T9 (GPR code 25) — R_TapePtr, the tape instruction stream pointer.
|
-- R_K0/K1 (codes 26-27) — kernel / interrupt handler reserves. Never touched by user code.
|
||||||
--- Owned by the tape runtime (in tape_run / tape_run_a02_s07).
|
-- R_GP/SP/FP/RA (codes 28-31) — R_SP/R_FP/R_RA are tape-runtime carriers between
|
||||||
--- `rgcc(R_TapePtr)` register-variable ties the C compiler's view to $t9 across the whole tape_run.
|
-- tape_enter and tape_exit; R_GP stays the host global pointer.
|
||||||
--- The auto-reg pass MUST NOT clobber this; doing so would desync the C-side tape pointer from the
|
|
||||||
--- hardware pointer and crash on the next tape_run.
|
|
||||||
---
|
|
||||||
--- R_T8 (GPR code 24) — R_AtomJmp, the atom-jump register used by the 4-word yield handshake.
|
|
||||||
--- Every `mac_yield()` / `mac_yield_tail` does `load_word R_AtomJmp, R_TapePtr, 0` then
|
|
||||||
--- `jump_reg R_AtomJmp`. The auto-reg pass MUST NOT clobber this either, or the atom dispatcher breaks.
|
|
||||||
--- Owned by the tape runtime, same family as R_TapePtr.
|
|
||||||
---
|
|
||||||
--- R_AT (GPR code 1) — Assembler temporary. Reserved by the MIPS O32 ABI for pseudoinstruction expansion
|
|
||||||
--- (lottes_tape.h:86, mips.h:93). The ISA's psuedo instructions use it as a scratch temporary.
|
|
||||||
---
|
|
||||||
--- R_A0..A3 (codes 4..7) — Function arguments. Used in tape_run_a02_s07, see below.
|
|
||||||
--- R_S0..S7 (codes 16..23) — Callee-saved. Preserved across C-ABI calls by convention.
|
|
||||||
--- The `tape_run_a02_s07` variant clobbers them deliberately, but the default `tape_run` does NOT.
|
|
||||||
--- Kept out of POOL to preserve the conservative default.
|
|
||||||
--- Add them in a separate "big clobber" pool if/when needed.
|
|
||||||
---
|
|
||||||
--- R_K0/K1 (codes 26..27) — Kernel / interrupt handler reserves. Never touched by user code; OS-internal.
|
|
||||||
--- R_GP/SP/FP/RA (codes 28..31) — Stack frame + return-address. Owned by the C compiler; never allocatable.
|
|
||||||
--- R_0 (code 0) — Hardwired zero. Cannot be written.
|
|
||||||
---
|
---
|
||||||
local POOL = {
|
local POOL = {
|
||||||
"R_T0", "R_T1", "R_T2", "R_T3",
|
"R_T0", "R_T1", "R_T2", "R_T3",
|
||||||
"R_T4", "R_T5", "R_T6", "R_T7",
|
"R_T4", "R_T5", "R_T6", "R_T7",
|
||||||
"R_V0", "R_V1",
|
"R_V0", "R_V1",
|
||||||
|
"R_A0", "R_A1", "R_A2", "R_A3",
|
||||||
|
"R_S0", "R_S1", "R_S2", "R_S3",
|
||||||
|
"R_S4", "R_S5", "R_S6", "R_S7",
|
||||||
|
"R_T8", "R_T9",
|
||||||
}
|
}
|
||||||
|
|
||||||
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
|
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
|
||||||
@@ -80,8 +64,12 @@ local POOL = {
|
|||||||
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
|
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
|
||||||
local INT_CODE_TO_POOL_GPR = {
|
local INT_CODE_TO_POOL_GPR = {
|
||||||
[2] = "R_V0", [3] = "R_V1",
|
[2] = "R_V0", [3] = "R_V1",
|
||||||
|
[4] = "R_A0", [5] = "R_A1", [6] = "R_A2", [7] = "R_A3",
|
||||||
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
||||||
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
||||||
|
[16] = "R_S0", [17] = "R_S1", [18] = "R_S2", [19] = "R_S3",
|
||||||
|
[20] = "R_S4", [21] = "R_S5", [22] = "R_S6", [23] = "R_S7",
|
||||||
|
[24] = "R_T8", [25] = "R_T9",
|
||||||
}
|
}
|
||||||
|
|
||||||
-- Stable sort for deterministic allocation order.
|
-- Stable sort for deterministic allocation order.
|
||||||
@@ -109,7 +97,7 @@ local function allocate_phase(phase_label, decls)
|
|||||||
line = 0,
|
line = 0,
|
||||||
msg = string.format("phase_register_pool_exhausted: "
|
msg = string.format("phase_register_pool_exhausted: "
|
||||||
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
|
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
|
||||||
.. "(max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
|
.. "(max 24 per phase: R_T0..R_T7 + R_V0..R_V1 + R_A0..R_A3 + R_S0..R_S7 + R_T8..R_T9). Split the phase or use hardcoded GPRs."
|
||||||
, phase_label, sym),
|
, phase_label, sym),
|
||||||
}
|
}
|
||||||
return result, errors
|
return result, errors
|
||||||
|
|||||||
+174
-18
@@ -7,7 +7,7 @@
|
|||||||
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||||
---
|
---
|
||||||
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
|
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
|
||||||
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
|
--- the ELF symbol is the C ident. Raw `MipsCode code_*` is leftover, not the atom rule.
|
||||||
---
|
---
|
||||||
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||||
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
||||||
@@ -130,6 +130,60 @@ local function extract_arg_names(args_str)
|
|||||||
for _, tok in ipairs(tokens) do
|
for _, tok in ipairs(tokens) do
|
||||||
local trimmed = duffle.trim(tok)
|
local trimmed = duffle.trim(tok)
|
||||||
if trimmed ~= "" then
|
if trimmed ~= "" then
|
||||||
|
-- Strip trailing block comment (/* ... */) from the token, if present.
|
||||||
|
-- split_top_level_commas only skips block comments at TOP LEVEL (between commas),
|
||||||
|
-- not block comments embedded WITHIN a token between a parameter and a trailing comma.
|
||||||
|
-- Without this strip, the identifier-walk below stops at the `/` of `*/` and returns
|
||||||
|
-- the wrong name (or nothing). See `test_extract_arg_names_handles_trailing_block_comments`.
|
||||||
|
local trimmed_end = #trimmed
|
||||||
|
if trimmed_end >= 2 and trimmed:sub(trimmed_end - 1, trimmed_end) == "*/" then
|
||||||
|
-- Find the matching `/*` that opens the trailing comment.
|
||||||
|
-- Walk back from the `*/` looking for `/*` (whitespace + `/*`).
|
||||||
|
local close_pos = trimmed_end - 1 -- position of the second-to-last char
|
||||||
|
-- Walk back: skip trailing whitespace, then look for the `/*` opener.
|
||||||
|
while close_pos > 1 do
|
||||||
|
local ch = trimmed:sub(close_pos, close_pos)
|
||||||
|
if ch == " " or ch == "\t" or ch == "\n" or ch == "\r" then
|
||||||
|
close_pos = close_pos - 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
-- Now scan back from close_pos for the `/*` opener (slashes are at close_pos-1 and close_pos-2).
|
||||||
|
local opener_pos = nil
|
||||||
|
local scan = close_pos - 3
|
||||||
|
while scan >= 1 do
|
||||||
|
if trimmed:sub(scan, scan + 1) == "/*" then
|
||||||
|
opener_pos = scan
|
||||||
|
break
|
||||||
|
end
|
||||||
|
scan = scan - 1
|
||||||
|
end
|
||||||
|
if opener_pos then
|
||||||
|
-- Truncate everything from opener_pos onwards.
|
||||||
|
trimmed = duffle.trim(trimmed:sub(1, opener_pos - 1))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if trimmed == "" then goto continue end
|
||||||
|
-- Strip trailing array suffix `[N]` if present.
|
||||||
|
-- Example: `Reg r_data[4]` → identifier is `r_data`, not `4`.
|
||||||
|
trimmed_end = #trimmed
|
||||||
|
if trimmed_end >= 4 and trimmed:sub(trimmed_end, trimmed_end) == "]" then
|
||||||
|
-- Walk back: skip digits, expect `[`.
|
||||||
|
local bracket_pos = trimmed_end - 1
|
||||||
|
while bracket_pos > 1 do
|
||||||
|
local ch = trimmed:sub(bracket_pos, bracket_pos)
|
||||||
|
if ch >= "0" and ch <= "9" then
|
||||||
|
bracket_pos = bracket_pos - 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if bracket_pos >= 1 and trimmed:sub(bracket_pos, bracket_pos) == "[" then
|
||||||
|
trimmed = duffle.trim(trimmed:sub(1, bracket_pos - 1))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if trimmed == "" then goto continue end
|
||||||
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
|
-- Find the identifier at the end: walk back over trailers (whitespace + `*` + `[]`),
|
||||||
-- then walk back over the identifier chars (alnum + `_`).
|
-- then walk back over the identifier chars (alnum + `_`).
|
||||||
local ident_end = #trimmed
|
local ident_end = #trimmed
|
||||||
@@ -153,12 +207,21 @@ local function extract_arg_names(args_str)
|
|||||||
ident_start = ident_start + 1
|
ident_start = ident_start + 1
|
||||||
local name = trimmed:sub(ident_start, ident_end)
|
local name = trimmed:sub(ident_start, ident_end)
|
||||||
if name ~= "" then names[#names + 1] = name end
|
if name ~= "" then names[#names + 1] = name end
|
||||||
|
::continue::
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
if #names == 0 then return nil end
|
if #names == 0 then return nil end
|
||||||
return names
|
return names
|
||||||
end
|
end
|
||||||
|
|
||||||
|
local function formal_arg_names(args_str)
|
||||||
|
local names = extract_arg_names(args_str)
|
||||||
|
if not names then return nil end
|
||||||
|
if names[1] == "ab" then table.remove(names, 1) end
|
||||||
|
if #names == 0 then return nil end
|
||||||
|
return names
|
||||||
|
end
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Component projection (read from pre-scanned SourceScan)
|
-- Component projection (read from pre-scanned SourceScan)
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -179,7 +242,7 @@ local function project_components(source, scan)
|
|||||||
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
|
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
|
||||||
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
|
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
|
||||||
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
|
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
|
||||||
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
|
-- `tb_emit` of the C ident, NOT inlined as a macro. Including `atom_proc` here
|
||||||
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
|
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
|
||||||
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
|
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
|
||||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||||
@@ -197,6 +260,7 @@ local function project_components(source, scan)
|
|||||||
body_off = a.body_off,
|
body_off = a.body_off,
|
||||||
body_tokens = a.body_tokens,
|
body_tokens = a.body_tokens,
|
||||||
args = args,
|
args = args,
|
||||||
|
arg_names = formal_arg_names(args),
|
||||||
comment = comment,
|
comment = comment,
|
||||||
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
kind = a.kind, -- "comp_bare" | "comp_proc"; provenance emitter reads this.
|
||||||
debug_skip = a.debug_skip == true,
|
debug_skip = a.debug_skip == true,
|
||||||
@@ -363,8 +427,10 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
|
|||||||
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
||||||
n = n + cycle_cost_rec(nested, comp_by_name, latency, cache)
|
n = n + cycle_cost_rec(nested, comp_by_name, latency, cache)
|
||||||
else
|
else
|
||||||
-- Leaf instruction or pseudo-macro. Look up in INSTRUCTION_LATENCY; default 1.
|
-- Leaf instruction or pseudo-macro.
|
||||||
n = n + (latency[ident] or 1)
|
local isa = duffle.instr(ident)
|
||||||
|
local gte = duffle.gte(ident)
|
||||||
|
n = n + ((isa and isa.cycles) or (gte and gte.cycles) or latency[ident] or 1)
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -384,6 +450,10 @@ end
|
|||||||
--- @param cache table<string, integer>
|
--- @param cache table<string, integer>
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function gp0_contrib_rec(name, comp_by_name, cache)
|
local function gp0_contrib_rec(name, comp_by_name, cache)
|
||||||
|
if name:match("^insert_ot_tag") then
|
||||||
|
cache[name] = 0
|
||||||
|
return 0
|
||||||
|
end
|
||||||
if cache[name] ~= nil then return cache[name] end
|
if cache[name] ~= nil then return cache[name] end
|
||||||
cache[name] = -1
|
cache[name] = -1
|
||||||
local cc = comp_by_name[name]
|
local cc = comp_by_name[name]
|
||||||
@@ -399,8 +469,15 @@ local function gp0_contrib_rec(name, comp_by_name, cache)
|
|||||||
-- Nested `mac_X(...)` call: recurse.
|
-- Nested `mac_X(...)` call: recurse.
|
||||||
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
||||||
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
||||||
|
elseif ident == "gte_sw" then
|
||||||
|
n = n + 1
|
||||||
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
|
elseif ident == "store_word" or ident == "store_half" or ident == "store_byte" then
|
||||||
if trimmed:find("R_PrimCursor", 1, true) then
|
if trimmed:find("R_PrimCursor", 1, true)
|
||||||
|
or trimmed:find("O_(Poly_", 1, true)
|
||||||
|
or trimmed:find("r_prim_cursor", 1, true)
|
||||||
|
or trimmed:find("r_primitive_cursor", 1, true)
|
||||||
|
or trimmed:find("r_base", 1, true)
|
||||||
|
then
|
||||||
n = n + 1
|
n = n + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -466,18 +543,9 @@ end
|
|||||||
--- @param args_str string|nil
|
--- @param args_str string|nil
|
||||||
--- @return string
|
--- @return string
|
||||||
local function signature_from_args(args_str)
|
local function signature_from_args(args_str)
|
||||||
local arg_names = extract_arg_names(args_str)
|
local names = formal_arg_names(args_str)
|
||||||
if arg_names and #arg_names > 0 then
|
if names then
|
||||||
-- Drop the leading `ab` (atom-builder) first arg if present.
|
return table.concat(names, ", ")
|
||||||
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
|
|
||||||
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
|
|
||||||
if arg_names[1] == "ab" then
|
|
||||||
table.remove(arg_names, 1)
|
|
||||||
end
|
|
||||||
if #arg_names > 0 then
|
|
||||||
return table.concat(arg_names, ", ")
|
|
||||||
end
|
|
||||||
return "..." -- `ab` was the only arg; fall through to variadic
|
|
||||||
end
|
end
|
||||||
return "..."
|
return "..."
|
||||||
end
|
end
|
||||||
@@ -491,16 +559,103 @@ local function strip_trailing_continuation(lines)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
--- Classify a token as a "pure delay marker token" (a delay-marker identifier
|
||||||
|
--- with no following instruction — only whitespace and/or block comments).
|
||||||
|
--- Examples that match:
|
||||||
|
--- * `GteDelay_` → marker alone
|
||||||
|
--- * `GteDelay_ /* RT diagonal: D1 = a.x... */` → marker + block comment
|
||||||
|
--- * `GteDelay_ /* RT diagonal: ... */\n\t` → marker + comment + trailing whitespace
|
||||||
|
--- Examples that DO NOT match (these contain a real instruction after the marker
|
||||||
|
--- and must be preserved verbatim so the instruction still gets emitted):
|
||||||
|
--- * `GteDelay_ nop2`
|
||||||
|
--- * `GteDelay_ add_si(r.dst_ptr, r.scratch, dst_offset)`
|
||||||
|
---
|
||||||
|
--- Why this classification matters: the metaprogram emits tokens separated by `,`
|
||||||
|
--- and joins them with `\<newline>` line continuations. After C preprocessor
|
||||||
|
--- phase 2 (line splicing), the macro body collapses to a single logical line.
|
||||||
|
--- Each delay-marker identifier expands to empty (its definition
|
||||||
|
--- `#define GteDelay_ // ...` consumes the `//` line comment during preprocessing
|
||||||
|
--- of the definition itself, leaving an empty replacement list). When a token
|
||||||
|
--- is purely a delay marker with only a trailing comment, the `,` the metaprogram
|
||||||
|
--- normally adds before each token-after-the-first brackets empty content and
|
||||||
|
--- produces the syntax error `,,` (`expected expression before ',' token`) at
|
||||||
|
--- C compile. The metaprogram therefore emits such tokens WITHOUT the leading
|
||||||
|
--- `,` (see `token_skips_leading_comma`) — but the marker + trailing comment
|
||||||
|
--- are still emitted verbatim so the annotation is preserved in `gen/macs.h`.
|
||||||
|
--- @param tok string -- a single token from split_top_level_commas (already trimmed at the start, may contain trailing whitespace + block comment)
|
||||||
|
--- @return boolean
|
||||||
|
local function is_pure_delay_marker_token(tok)
|
||||||
|
local markers = duffle.DELAY_MARKERS
|
||||||
|
if type(markers) ~= "table" then return false end
|
||||||
|
|
||||||
|
-- Identify a leading delay-marker identifier (e.g. `GteDelay_`).
|
||||||
|
local ident_end = 1
|
||||||
|
while ident_end <= #tok do
|
||||||
|
local ch = tok:sub(ident_end, ident_end)
|
||||||
|
if ch:match("[%w_]") then
|
||||||
|
ident_end = ident_end + 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local ident = tok:sub(1, ident_end - 1)
|
||||||
|
if not markers[ident] then return false end
|
||||||
|
|
||||||
|
-- Walk the remainder: only whitespace and block comments are allowed.
|
||||||
|
local scan = ident_end
|
||||||
|
while scan <= #tok do
|
||||||
|
local ch = tok:sub(scan, scan)
|
||||||
|
if ch:match("%s") then
|
||||||
|
scan = scan + 1
|
||||||
|
elseif ch == "/" and tok:sub(scan + 1, scan + 1) == "*" then
|
||||||
|
local close = tok:find("*/", scan + 2, true)
|
||||||
|
if not close then return false end
|
||||||
|
scan = close + 2
|
||||||
|
else
|
||||||
|
-- Non-whitespace, non-block-comment content: a real instruction
|
||||||
|
-- follows the marker (e.g. `GteDelay_ nop2`); keep this token intact.
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return true
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Classify a token's "leading comma requirement".
|
||||||
|
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_`
|
||||||
|
--- followed by whitespace + optional block comment and NOTHING ELSE) expand
|
||||||
|
--- to empty at C preprocessor time. Emitting them WITHOUT the leading `,`
|
||||||
|
--- separator that the metaprogram normally adds before each token after the
|
||||||
|
--- first keeps exactly one `,` between the surrounding real expressions in
|
||||||
|
--- the spliced macro body:
|
||||||
|
---
|
||||||
|
--- * before this rule: `<tok1> ,\t<gdelay> ,\t<tok3>` → after expansion
|
||||||
|
--- `<tok1> , /* comment */ , <tok3>` → `,,` syntax error.
|
||||||
|
--- * after this rule: `<tok1> \t<gdelay> ,\t<tok3>` → after expansion
|
||||||
|
--- `<tok1> /* comment */ , <tok3>` → `<tok1>, <tok3>` — valid.
|
||||||
|
---
|
||||||
|
--- Tokens like `GteDelay_ nop2` keep the leading `,` (the marker is followed
|
||||||
|
--- by a real instruction, so the marker + instruction together need the
|
||||||
|
--- separator on the LEFT to land between two real expressions).
|
||||||
|
--- @param tok string
|
||||||
|
--- @return boolean -- true if the token needs NO leading `,` separator.
|
||||||
|
local function token_skips_leading_comma(tok)
|
||||||
|
return is_pure_delay_marker_token(tok)
|
||||||
|
end
|
||||||
|
|
||||||
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
|
--- Emit the `#define mac_X(sig) \<newline>\t<tok1> \<newline>,\t<tok2> ...` block.
|
||||||
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
|
--- Converts `//` line comments to `/* */` block comments in each token so they don't break the C macro `\` line continuations.
|
||||||
|
---
|
||||||
|
--- Pure delay-marker tokens (`GteDelay_` / `LdSlot_` / `BdSlot_` / `DmaSlot_` with only a trailing block comment, no real instruction) are emitted WITHOUT a leading `,` separator; the annotation IS preserved in the generated header (so the comment + marker remain visible to anyone reading `gen/macs.h`), but the C preprocessor expands the marker to empty, so leaving the `,` separator out is what stops the `,,` syntax error. See `token_skips_leading_comma` for the contract.
|
||||||
local function emit_macro_body(lines, c, sig, tokens)
|
local function emit_macro_body(lines, c, sig, tokens)
|
||||||
for tok_idx = 1, #tokens do
|
for tok_idx = 1, #tokens do
|
||||||
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
|
tokens[tok_idx] = convert_line_comments_to_block(tokens[tok_idx])
|
||||||
end
|
end
|
||||||
|
if #tokens == 0 then return end
|
||||||
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
|
lines[#lines + 1] = "#define mac_" .. c.name .. "(" .. sig .. ") \\"
|
||||||
lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
|
lines[#lines + 1] = "\t" .. tokens[1] .. " \\"
|
||||||
for tok_idx = 2, #tokens do
|
for tok_idx = 2, #tokens do
|
||||||
lines[#lines + 1] = ",\t" .. tokens[tok_idx] .. " \\"
|
local sep = token_skips_leading_comma(tokens[tok_idx]) and "\t" or ",\t"
|
||||||
|
lines[#lines + 1] = sep .. tokens[tok_idx] .. " \\"
|
||||||
end
|
end
|
||||||
strip_trailing_continuation(lines)
|
strip_trailing_continuation(lines)
|
||||||
end
|
end
|
||||||
@@ -710,6 +865,7 @@ local function update_canonical_component_body_index(corpus, src, components, sc
|
|||||||
source = src.path,
|
source = src.path,
|
||||||
declaration = c.line,
|
declaration = c.line,
|
||||||
kind = c.kind,
|
kind = c.kind,
|
||||||
|
arg_names = c.arg_names,
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
---
|
---
|
||||||
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
|
--- Reads the post-link ELF directly (io.open; walks the ELF32 section header table to find
|
||||||
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
|
--- `.debug_info` + `.debug_abbrev` + `.debug_str` + `.debug_line` + `.debug_aranges` + `.debug_rnglists`),
|
||||||
--- APPENDS synthetic DWARF line-program sequences for every `code_<name>` atom, EXTENDS the `.debug_aranges`
|
--- APPENDS synthetic DWARF line-program sequences for every tape atom, EXTENDS the `.debug_aranges`
|
||||||
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
|
--- and main-CU range tables with the atom ranges, and INSERTS synthetic atom/component DIE children into the
|
||||||
--- existing main compilation unit in `.debug_info` (no second compilation unit).
|
--- existing main compilation unit in `.debug_info` (no second compilation unit).
|
||||||
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
|
--- Per-atom `DW_TAG_subprogram` + per-register `DW_TAG_variable` entries make
|
||||||
@@ -1344,7 +1344,7 @@ local function build_new_abbrev()
|
|||||||
attr( DW_AT_name, DW_FORM_string)
|
attr( DW_AT_name, DW_FORM_string)
|
||||||
.. attr(DW_AT_low_pc, DW_FORM_addr)
|
.. attr(DW_AT_low_pc, DW_FORM_addr)
|
||||||
.. attr(DW_AT_high_pc, DW_FORM_addr)
|
.. attr(DW_AT_high_pc, DW_FORM_addr)
|
||||||
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; lets gdb's symbol-table lookup resolve to our subprogram (not the gcc global `code_<name>` const U4 array)
|
.. attr(DW_AT_linkage_name, DW_FORM_string)) -- equals DW_AT_name; gdb resolves the subprogram, not the gcc global array
|
||||||
|
|
||||||
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
|
local abbrev_variable = abbrev(ABBREV_VARIABLE, DW_TAG_variable, false, -- DW_CHILDREN_no
|
||||||
attr( DW_AT_name, DW_FORM_string)
|
attr( DW_AT_name, DW_FORM_string)
|
||||||
@@ -1857,7 +1857,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
|
-- 4) Emit per-atom DW_TAG_subprograms (children of main CU).
|
||||||
-- Subprogram names match nm symbols without a `code_` prefix.
|
-- Subprogram names match the written C ident (the ELF symbol).
|
||||||
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
|
-- The gcc global `<name>[]` is a DW_TAG_variable without children; our subprogram has the wave-context var children.
|
||||||
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
|
-- gdb's symbol resolution picks our subprogram (it has low_pc/high_pc + children) over the gcc global for function-context lookups.
|
||||||
for _, atom in ipairs(atom_table) do
|
for _, atom in ipairs(atom_table) do
|
||||||
@@ -2313,5 +2313,6 @@ end
|
|||||||
M.compute_loclists_offsets_for_test = compute_loclists_offsets
|
M.compute_loclists_offsets_for_test = compute_loclists_offsets
|
||||||
M.build_debug_loclists_section_for_test = build_debug_loclists_section
|
M.build_debug_loclists_section_for_test = build_debug_loclists_section
|
||||||
M.tape_piece_size_for_test = tape_piece_size
|
M.tape_piece_size_for_test = tape_piece_size
|
||||||
|
M.build_atom_table_for_test = build_atom_table
|
||||||
|
|
||||||
return M
|
return M
|
||||||
|
|||||||
@@ -155,8 +155,28 @@ local function project_atom(atom_record, src, corpus)
|
|||||||
local body = atom_record.body or ""
|
local body = atom_record.body or ""
|
||||||
local wc = corpus.word_counts or {}
|
local wc = corpus.word_counts or {}
|
||||||
local cbi = corpus.component_body_index or {}
|
local cbi = corpus.component_body_index or {}
|
||||||
|
local schema = nil
|
||||||
|
if atom_record.reg_use_schema_name then
|
||||||
|
schema = corpus.reg_use_schemas and corpus.reg_use_schemas[atom_record.reg_use_schema_name]
|
||||||
|
end
|
||||||
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
-- That construction site stamps `invocation.debug_skip` while appending each record to `proj.invocations`.
|
||||||
local proj = duffle.project_emission(body, cbi, wc, corpus.components)
|
local proj = duffle.project_emission(body, cbi, wc, corpus.components, {
|
||||||
|
reg_use_schema = schema,
|
||||||
|
reg_use_param = atom_record.reg_use_param_name,
|
||||||
|
atom_name = atom_record.name,
|
||||||
|
schema_name = atom_record.reg_use_schema_name,
|
||||||
|
})
|
||||||
|
if atom_record.reg_use_schema_name and not schema then
|
||||||
|
proj.errors[#proj.errors + 1] = {
|
||||||
|
kind = "reguse_missing_schema",
|
||||||
|
msg = string.format("RegUse schema %q is missing", atom_record.reg_use_schema_name),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
for _, err in ipairs(corpus.reg_use_errors or {}) do
|
||||||
|
if err.schema_name == atom_record.reg_use_schema_name then
|
||||||
|
proj.errors[#proj.errors + 1] = err
|
||||||
|
end
|
||||||
|
end
|
||||||
local paths = {
|
local paths = {
|
||||||
tokens = atom_record.body_tokens or {},
|
tokens = atom_record.body_tokens or {},
|
||||||
line_in_body = duffle.build_body_line_index(body),
|
line_in_body = duffle.build_body_line_index(body),
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
--- passes/offsets.lua — Branch-offset generator.
|
--- passes/offsets.lua — Branch-offset generator.
|
||||||
---
|
---
|
||||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||||
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
--- for `MipsAtom_(name)` and leftover `MipsCode code_*` declarations, computes the word offset
|
||||||
|
--- (ELF symbol is the C ident; raw `code_*` is leftover, not the atom rule)
|
||||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||||
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||||
---
|
---
|
||||||
|
|||||||
+578
-227
@@ -4,8 +4,8 @@
|
|||||||
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
||||||
--- - `build/gen/annotation_validation.txt` — the project summary.
|
--- - `build/gen/annotation_validation.txt` — the project summary.
|
||||||
---
|
---
|
||||||
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
|
--- The canonical `corpus.sources_by_dir` projection groups sources by directory.
|
||||||
--- This pass iterates the dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
|
--- This pass builds one ModuleView per directory and walks SECTION_RENDERERS.
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Module-scope requires + package.path setup
|
-- Module-scope requires + package.path setup
|
||||||
@@ -20,11 +20,6 @@
|
|||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
|
||||||
-- Load the annotation pass so we can re-validate each source against the canonical corpus projection.
|
|
||||||
-- The annotation pass exposes `M.validate`, which returns the per-source AnnotationResult (atoms / annots / macros / binds / errors / warnings)
|
|
||||||
-- that the report pass renders into the per-module `<dir_basename>.annotations.txt` output.
|
|
||||||
local annotation = dofile(_bootstrap_dir .. "annotation.lua")
|
|
||||||
|
|
||||||
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
|
-- Load atoms_source_map for the `render_source_map` / `render_provenance` module functions (used by `render_module_atoms_md` to produce `<module>.atoms.md` without re-walking source tokens).
|
||||||
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
|
-- The pass itself emits no per-source files anymore; we only consume the two pure renderers here.
|
||||||
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
|
-- Defined BEFORE the renderer functions below so their upvalues resolve to this local (not the global `atoms_source_map`, which is nil).
|
||||||
@@ -225,7 +220,7 @@ local function render_module_atoms_md(dir, dir_sources, wc)
|
|||||||
for _, atom in ipairs(atoms_list) do
|
for _, atom in ipairs(atoms_list) do
|
||||||
lines[#lines + 1] = string.format(
|
lines[#lines + 1] = string.format(
|
||||||
"### atom: %s (line %d, %d words)",
|
"### atom: %s (line %d, %d words)",
|
||||||
atom.name, atom.line or 0, #(atom.paths.items or {}))
|
atom.name, atom.line or 0, #((atom.paths or {}).word_events or {}))
|
||||||
lines[#lines + 1] = ""
|
lines[#lines + 1] = ""
|
||||||
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
|
lines[#lines + 1] = "**Sourcemap** — per-word call site:"
|
||||||
lines[#lines + 1] = "```"
|
lines[#lines + 1] = "```"
|
||||||
@@ -246,17 +241,544 @@ local function render_module_atoms_md(dir, dir_sources, wc)
|
|||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
|
local function decl_words(atom)
|
||||||
--- Aggregates annotation + static-analysis content across all sources in `dir`.
|
local p = atom.paths or {}
|
||||||
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
|
return #(p.word_events or {})
|
||||||
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
|
end
|
||||||
--- @param dir string
|
|
||||||
--- @param dir_sources SourceFile[]
|
local function count_kinds(decls)
|
||||||
--- @param annot_results AnnotationResult[]
|
local n = { atom = 0, atom_proc = 0, comp_bare = 0, comp_proc = 0 }
|
||||||
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
|
for _, a in ipairs(decls or {}) do
|
||||||
--- @return string
|
if n[a.kind] ~= nil then n[a.kind] = n[a.kind] + 1 end
|
||||||
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
end
|
||||||
|
return n
|
||||||
|
end
|
||||||
|
|
||||||
|
local function slot_suffix(key)
|
||||||
|
if type(key) ~= "string" or key:sub(1, 7) ~= "reguse:" then return nil end
|
||||||
|
return key:match("([^:]+)$")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function decl_names(view)
|
||||||
|
local names = {}
|
||||||
|
for _, a in ipairs(view.decls or {}) do
|
||||||
|
if a.name then names[a.name] = true end
|
||||||
|
end
|
||||||
|
return names
|
||||||
|
end
|
||||||
|
|
||||||
|
local function path_in_module(path, view)
|
||||||
|
if type(path) ~= "string" or path == "" then return false end
|
||||||
|
local norm = path:gsub("\\", "/")
|
||||||
|
local dir = (view.dir or ""):gsub("\\", "/")
|
||||||
|
if dir ~= "" and (norm == dir or norm:sub(1, #dir + 1) == dir .. "/") then
|
||||||
|
return true
|
||||||
|
end
|
||||||
|
for _, src in ipairs(view.sources or {}) do
|
||||||
|
if (src.path or ""):gsub("\\", "/") == norm then return true end
|
||||||
|
end
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
|
||||||
|
local function build_module_view(dir, dir_sources, corpus)
|
||||||
|
local decls = {}
|
||||||
|
for _, src in ipairs(dir_sources or {}) do
|
||||||
|
for _, a in ipairs((src.scan and src.scan.atoms) or {}) do
|
||||||
|
if not a.source_path then a.source_path = src.path end
|
||||||
|
decls[#decls + 1] = a
|
||||||
|
end
|
||||||
|
end
|
||||||
local dir_basename = source_basename(dir)
|
local dir_basename = source_basename(dir)
|
||||||
|
local sa = (corpus.static_analysis_results or {})[dir_basename] or {}
|
||||||
|
local schemas = {}
|
||||||
|
for name, schema in pairs(corpus.reg_use_schemas or {}) do
|
||||||
|
for _, a in ipairs(decls) do
|
||||||
|
if a.reg_use_schema_name == name then
|
||||||
|
schemas[#schemas + 1] = schema
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return {
|
||||||
|
dir = dir,
|
||||||
|
sources = dir_sources or {},
|
||||||
|
decls = decls,
|
||||||
|
schemas = schemas,
|
||||||
|
findings = sa.findings or {},
|
||||||
|
sa = sa,
|
||||||
|
corpus = corpus,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_declarations(add, view)
|
||||||
|
if #view.decls == 0 then add("_(none)_"); add(""); return end
|
||||||
|
add("| kind | name | source | line | words | min | max | branches | paths |")
|
||||||
|
add("|------|------|--------|------|-------|-----|-----|----------|-------|")
|
||||||
|
for _, a in ipairs(view.decls) do
|
||||||
|
local p = a.paths or {}
|
||||||
|
add(string.format("| %s | %s | %s | %d | %d | %s | %s | %s | %s |",
|
||||||
|
a.kind or "?",
|
||||||
|
a.name or "?",
|
||||||
|
source_basename(a.source_path or ""),
|
||||||
|
a.line or 0,
|
||||||
|
decl_words(a),
|
||||||
|
tostring(p.cycles_min or "—"),
|
||||||
|
tostring(p.cycles_max or "—"),
|
||||||
|
tostring(p.branches or "—"),
|
||||||
|
tostring(p.paths or "—")))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_components(add, view)
|
||||||
|
local rows = {}
|
||||||
|
local index = (view.corpus and view.corpus.component_body_index) or {}
|
||||||
|
for _, a in ipairs(view.decls) do
|
||||||
|
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||||
|
local idx = index[a.name] or {}
|
||||||
|
local args = idx.arg_names or {}
|
||||||
|
rows[#rows + 1] = {
|
||||||
|
name = a.name,
|
||||||
|
kind = a.kind,
|
||||||
|
args = table.concat(args, ", "),
|
||||||
|
words = decl_words(a),
|
||||||
|
map = a.map_command or "—",
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #rows == 0 then add("_(none)_"); add(""); return end
|
||||||
|
add("| name | kind | arg_names | words | map |")
|
||||||
|
add("|------|------|-----------|-------|-----|")
|
||||||
|
for _, r in ipairs(rows) do
|
||||||
|
add(string.format("| %s | %s | %s | %d | %s |",
|
||||||
|
r.name, r.kind, r.args ~= "" and r.args or "—", r.words, r.map))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_reguse(add, view)
|
||||||
|
local wrote = false
|
||||||
|
for _, schema in ipairs(view.schemas or {}) do
|
||||||
|
wrote = true
|
||||||
|
add(string.format("### %s", schema.name or "?"))
|
||||||
|
for _, slot in ipairs(schema.slots or {}) do
|
||||||
|
local aliases = table.concat(slot.aliases or { slot.name }, ", ")
|
||||||
|
local ro = slot.readonly and " readonly" or ""
|
||||||
|
add(string.format("- slot `%s` aliases %s%s", slot.name, aliases, ro))
|
||||||
|
end
|
||||||
|
for _, a in ipairs(view.decls) do
|
||||||
|
if a.reg_use_schema_name == schema.name then
|
||||||
|
add(string.format("- bound `%s` param `%s`", a.name, a.reg_use_param_name or "?"))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
local bound = {}
|
||||||
|
for _, schema in ipairs(view.schemas or {}) do
|
||||||
|
if schema.name then bound[schema.name] = true end
|
||||||
|
end
|
||||||
|
local errors = {}
|
||||||
|
for _, err in ipairs((view.corpus and view.corpus.reg_use_errors) or {}) do
|
||||||
|
if bound[err.schema_name] or path_in_module(err.source_file, view) then
|
||||||
|
errors[#errors + 1] = err
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #errors > 0 then
|
||||||
|
wrote = true
|
||||||
|
add("### parse errors")
|
||||||
|
for _, err in ipairs(errors) do
|
||||||
|
add(string.format("- `%s` %s", err.kind or "?", err.schema_name or ""))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
if not wrote then add("_(none)_"); add("") end
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_annotations(add, view)
|
||||||
|
local rows = {}
|
||||||
|
for _, src in ipairs(view.sources) do
|
||||||
|
for _, info in ipairs((src.scan and src.scan.atom_infos) or {}) do
|
||||||
|
rows[#rows + 1] = {
|
||||||
|
source = source_basename(src.path),
|
||||||
|
line = info.info_line or 0,
|
||||||
|
name = info.atom_name or "?",
|
||||||
|
binds = info.binds or "—",
|
||||||
|
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "—",
|
||||||
|
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "—",
|
||||||
|
phase = info.phase or "—",
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #rows == 0 then add("_(none)_"); add(""); return end
|
||||||
|
add("| source | line | name | binds | reads | writes | phase |")
|
||||||
|
add("|--------|------|------|-------|-------|--------|-------|")
|
||||||
|
for _, r in ipairs(rows) do
|
||||||
|
add(string.format("| %s | %d | %s | %s | %s | %s | %s |",
|
||||||
|
r.source, r.line, r.name, r.binds, r.reads, r.writes, r.phase))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_component_annotations(add, view)
|
||||||
|
local rows = {}
|
||||||
|
for _, src in ipairs(view.sources) do
|
||||||
|
for _, info in ipairs((src.scan and src.scan.component_atom_infos) or {}) do
|
||||||
|
rows[#rows + 1] = {
|
||||||
|
source = source_basename(src.path),
|
||||||
|
line = info.info_line or 0,
|
||||||
|
name = info.atom_name or "?",
|
||||||
|
reads = (#(info.reads or {}) > 0 and table.concat(info.reads, ",")) or "—",
|
||||||
|
writes = (#(info.writes or {}) > 0 and table.concat(info.writes, ",")) or "—",
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #rows == 0 then add("_(none)_"); add(""); return end
|
||||||
|
add("| source | line | name | reads | writes |")
|
||||||
|
add("|--------|------|------|-------|--------|")
|
||||||
|
for _, r in ipairs(rows) do
|
||||||
|
add(string.format("| %s | %d | %s | %s | %s |",
|
||||||
|
r.source, r.line, r.name, r.reads, r.writes))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_binds(add, view)
|
||||||
|
local wrote = false
|
||||||
|
for _, src in ipairs(view.sources) do
|
||||||
|
for _, b in ipairs((src.scan and src.scan.binds) or {}) do
|
||||||
|
wrote = true
|
||||||
|
add(string.format("### %s (%s:%s, %s bytes)",
|
||||||
|
b.name, source_basename(src.path), tostring(b.line or 0), tostring(b.bytes or "—")))
|
||||||
|
for _, f in ipairs(b.fields or {}) do
|
||||||
|
add(string.format("- `+%s %s`", tostring(f.offset or "?"), f.name or "?"))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if not wrote then add("_(none)_"); add("") end
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_phases(add, view)
|
||||||
|
local corpus = view.corpus or {}
|
||||||
|
local names = decl_names(view)
|
||||||
|
local wrote = false
|
||||||
|
for phase, entry in pairs(corpus.atom_phases or {}) do
|
||||||
|
local here = {}
|
||||||
|
for _, atom_name in ipairs(entry.atoms or {}) do
|
||||||
|
if names[atom_name] then here[#here + 1] = atom_name end
|
||||||
|
end
|
||||||
|
if #here > 0 then
|
||||||
|
wrote = true
|
||||||
|
add(string.format("- phase `%s`: %s", phase, table.concat(here, ", ")))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for name, entry in pairs(corpus.atom_views or {}) do
|
||||||
|
if names[name] then
|
||||||
|
wrote = true
|
||||||
|
add(string.format("- view `%s` binds `%s`", name, entry.binds_name or "—"))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for name, entry in pairs(corpus.atom_ctxs or {}) do
|
||||||
|
if names[name] then
|
||||||
|
wrote = true
|
||||||
|
add(string.format("- ctx `%s` rbind `%s`", name, entry.rbind_atom or "—"))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if not wrote then add("_(none)_") end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_aliases(add, view)
|
||||||
|
local names = {}
|
||||||
|
local seen = {}
|
||||||
|
for _, src in ipairs(view.sources or {}) do
|
||||||
|
for name, entry in pairs((src.scan and src.scan.register_alias_registry) or {}) do
|
||||||
|
if not seen[name] then
|
||||||
|
seen[name] = entry
|
||||||
|
names[#names + 1] = name
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
table.sort(names)
|
||||||
|
if #names == 0 then add("_(none)_"); add(""); return end
|
||||||
|
add("| alias | type |")
|
||||||
|
add("|-------|------|")
|
||||||
|
for _, name in ipairs(names) do
|
||||||
|
local e = seen[name]
|
||||||
|
add(string.format("| %s | %s |", name, (e and e.default_type) or "—"))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_autoreg(add, view)
|
||||||
|
local allowed = decl_names(view)
|
||||||
|
for phase, entry in pairs((view.corpus and view.corpus.atom_phases) or {}) do
|
||||||
|
for _, atom_name in ipairs(entry.atoms or {}) do
|
||||||
|
if allowed[atom_name] then allowed[phase] = true end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local wrote = false
|
||||||
|
local seen = {}
|
||||||
|
local function dump(label, table_map)
|
||||||
|
local scopes = {}
|
||||||
|
for scope in pairs(table_map or {}) do
|
||||||
|
if allowed[scope] and not seen[label .. "\0" .. scope] then
|
||||||
|
scopes[#scopes + 1] = scope
|
||||||
|
end
|
||||||
|
end
|
||||||
|
table.sort(scopes)
|
||||||
|
for _, scope in ipairs(scopes) do
|
||||||
|
seen[label .. "\0" .. scope] = true
|
||||||
|
wrote = true
|
||||||
|
local syms = {}
|
||||||
|
for sym, gpr in pairs(table_map[scope] or {}) do
|
||||||
|
if type(gpr) == "string" and gpr ~= sym then
|
||||||
|
syms[#syms + 1] = string.format("%s → %s", sym, gpr)
|
||||||
|
else
|
||||||
|
syms[#syms + 1] = tostring(sym)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
table.sort(syms)
|
||||||
|
add(string.format("- %s `%s`: %s", label, scope, table.concat(syms, ", ")))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local corpus = view.corpus or {}
|
||||||
|
dump("atom", corpus.atom_auto_regs)
|
||||||
|
dump("phase", corpus.phase_auto_regs)
|
||||||
|
for _, src in ipairs(view.sources or {}) do
|
||||||
|
dump("atom", src.scan and src.scan.atom_auto_regs)
|
||||||
|
dump("phase", src.scan and src.scan.phase_auto_regs)
|
||||||
|
end
|
||||||
|
if not wrote then add("_(none)_") end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_collisions(add, view)
|
||||||
|
local rows = {}
|
||||||
|
for _, c in ipairs((view.corpus and view.corpus.collisions) or {}) do
|
||||||
|
local first = c.first_site or {}
|
||||||
|
local other = c.conflicting_site or {}
|
||||||
|
if path_in_module(first.path, view) or path_in_module(other.path, view) then
|
||||||
|
rows[#rows + 1] = c
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #rows == 0 then add("_(none)_"); add(""); return end
|
||||||
|
for _, c in ipairs(rows) do
|
||||||
|
local first = c.first_site or {}
|
||||||
|
local other = c.conflicting_site or {}
|
||||||
|
add(string.format("- `%s` `%s` first %s:%s conflict %s:%s",
|
||||||
|
c.kind or "?", c.name or "?",
|
||||||
|
tostring(first.path or "?"), tostring(first.line or "?"),
|
||||||
|
tostring(other.path or "?"), tostring(other.line or "?")))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_findings(add, view)
|
||||||
|
local by_atom = {}
|
||||||
|
for _, f in ipairs(view.findings or {}) do
|
||||||
|
local key = f.atom or "?"
|
||||||
|
by_atom[key] = by_atom[key] or {}
|
||||||
|
by_atom[key][#by_atom[key] + 1] = f
|
||||||
|
end
|
||||||
|
if next(by_atom) == nil then add("_(none)_"); add(""); return end
|
||||||
|
local seen = {}
|
||||||
|
local function emit(name, fs)
|
||||||
|
add("### " .. name)
|
||||||
|
for _, f in ipairs(fs) do
|
||||||
|
local msg = f.msg or ""
|
||||||
|
local slot = slot_suffix(f.gpr_key or f.producer_destination)
|
||||||
|
if slot and not msg:find("(slot ", 1, true) then
|
||||||
|
msg = msg .. " (slot " .. slot .. ")"
|
||||||
|
end
|
||||||
|
add(string.format("- `[%s/%s] %s`", f.kind or "info", f.check or "?", msg))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
for _, a in ipairs(view.decls) do
|
||||||
|
if by_atom[a.name] then
|
||||||
|
seen[a.name] = true
|
||||||
|
emit(a.name, by_atom[a.name])
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local leftovers = {}
|
||||||
|
for name in pairs(by_atom) do
|
||||||
|
if not seen[name] then leftovers[#leftovers + 1] = name end
|
||||||
|
end
|
||||||
|
table.sort(leftovers)
|
||||||
|
for _, name in ipairs(leftovers) do emit(name, by_atom[name]) end
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_relations(add, view)
|
||||||
|
local wrote = false
|
||||||
|
for _, a in ipairs(view.decls) do
|
||||||
|
local rels = (a.paths and a.paths.relations) or {}
|
||||||
|
if #rels > 0 then
|
||||||
|
wrote = true
|
||||||
|
add("### " .. a.name)
|
||||||
|
for _, rel in ipairs(rels) do
|
||||||
|
local dest = rel.destination or rel.producer_destination or "—"
|
||||||
|
local slot = slot_suffix(dest)
|
||||||
|
local dest_s = tostring(dest)
|
||||||
|
if slot then dest_s = dest_s .. " (slot " .. slot .. ")" end
|
||||||
|
add(string.format("- `%s` words %s → %s dest %s",
|
||||||
|
rel.semantic or "?",
|
||||||
|
tostring(rel.producer_word or "?"),
|
||||||
|
tostring(rel.consumer_word or "?"),
|
||||||
|
dest_s))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if not wrote then add("_(none)_"); add("") end
|
||||||
|
end
|
||||||
|
|
||||||
|
local HIDDEN_UNLESS_WRITTEN = {
|
||||||
|
R_AT = true, R_TapePtr = true, R_AtomJmp = true,
|
||||||
|
}
|
||||||
|
|
||||||
|
local PHYSICAL_GPR = {
|
||||||
|
R_T0 = true, R_T1 = true, R_T2 = true, R_T3 = true,
|
||||||
|
R_T4 = true, R_T5 = true, R_T6 = true, R_T7 = true,
|
||||||
|
R_V0 = true, R_V1 = true,
|
||||||
|
}
|
||||||
|
|
||||||
|
local function encoder_wrote_key(atom, key)
|
||||||
|
for _, ev in ipairs((atom.paths and atom.paths.word_events) or {}) do
|
||||||
|
for _, dest in pairs(ev.gpr_keys or {}) do
|
||||||
|
if dest == key then return true end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
|
||||||
|
local function written_name_for(key, atom)
|
||||||
|
local slot = key:match("^reguse:.+:(.+)$")
|
||||||
|
if slot then
|
||||||
|
local param = atom.reg_use_param_name
|
||||||
|
if param and param ~= "" then return param .. "." .. slot end
|
||||||
|
return slot
|
||||||
|
end
|
||||||
|
return key
|
||||||
|
end
|
||||||
|
|
||||||
|
local function aliases_for_key(key, atom, view)
|
||||||
|
local slot = key:match("^reguse:.+:(.+)$")
|
||||||
|
if not slot then return "—" end
|
||||||
|
local schema_name = atom.reg_use_schema_name
|
||||||
|
local schema = view.corpus and view.corpus.reg_use_schemas and view.corpus.reg_use_schemas[schema_name]
|
||||||
|
if not schema then return "—" end
|
||||||
|
for _, s in ipairs(schema.slots or {}) do
|
||||||
|
if s.name == slot then
|
||||||
|
local names = {}
|
||||||
|
for _, alias in ipairs(s.aliases or {}) do
|
||||||
|
if alias ~= slot then names[#names + 1] = alias end
|
||||||
|
end
|
||||||
|
if #names == 0 then
|
||||||
|
if s.aliases and #s.aliases > 0 then return table.concat(s.aliases, ", ") end
|
||||||
|
return "—"
|
||||||
|
end
|
||||||
|
return table.concat(names, ", ")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return "—"
|
||||||
|
end
|
||||||
|
|
||||||
|
local function physical_for_key(key, atom, view)
|
||||||
|
if PHYSICAL_GPR[key] then return key end
|
||||||
|
local corpus = view.corpus or {}
|
||||||
|
local alias = (corpus.register_alias_registry or {})[key]
|
||||||
|
if type(alias) == "table" then
|
||||||
|
local phys = alias.physical or alias.gpr or alias.code_name
|
||||||
|
if type(phys) == "string" and PHYSICAL_GPR[phys] then return phys end
|
||||||
|
if type(alias.name) == "string" and PHYSICAL_GPR[alias.name] then return alias.name end
|
||||||
|
elseif type(alias) == "string" and PHYSICAL_GPR[alias] then
|
||||||
|
return alias
|
||||||
|
end
|
||||||
|
local atom_map = (corpus.atom_auto_regs or {})[atom.name]
|
||||||
|
if type(atom_map) == "table" then
|
||||||
|
local slot = key:match("^reguse:.+:(.+)$") or key
|
||||||
|
local bound = atom_map[slot] or atom_map["R_" .. slot]
|
||||||
|
if type(bound) == "string" and PHYSICAL_GPR[bound] then return bound end
|
||||||
|
end
|
||||||
|
return "—"
|
||||||
|
end
|
||||||
|
|
||||||
|
local function last_relation_for(key, atom)
|
||||||
|
local last = nil
|
||||||
|
for _, rel in ipairs((atom.paths and atom.paths.relations) or {}) do
|
||||||
|
local dest = rel.destination or rel.producer_destination
|
||||||
|
if dest == key then last = rel end
|
||||||
|
end
|
||||||
|
if not last then return "—" end
|
||||||
|
local sem = last.semantic or "?"
|
||||||
|
local a = last.producer_word
|
||||||
|
local b = last.consumer_word
|
||||||
|
if a and b then return string.format("%s w%s→%s", sem, tostring(a), tostring(b)) end
|
||||||
|
return sem
|
||||||
|
end
|
||||||
|
|
||||||
|
local function render_section_forward(add, view)
|
||||||
|
local wrote = false
|
||||||
|
for _, a in ipairs(view.decls) do
|
||||||
|
local gpr = a.paths and a.paths.forward_state and a.paths.forward_state.gpr_values
|
||||||
|
local keys = {}
|
||||||
|
for k in pairs(gpr or {}) do
|
||||||
|
if k == "R_0" then
|
||||||
|
-- hidden
|
||||||
|
elseif HIDDEN_UNLESS_WRITTEN[k] and not encoder_wrote_key(a, k) then
|
||||||
|
-- hidden
|
||||||
|
else
|
||||||
|
keys[#keys + 1] = k
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #keys > 0 then
|
||||||
|
wrote = true
|
||||||
|
add("### " .. a.name)
|
||||||
|
add("| written | aliases | physical | lattice | last relation |")
|
||||||
|
add("|---|---|---|---|---|")
|
||||||
|
table.sort(keys)
|
||||||
|
for _, k in ipairs(keys) do
|
||||||
|
local slot = gpr[k]
|
||||||
|
local lattice = "—"
|
||||||
|
if slot and slot.kind == "constant" then
|
||||||
|
lattice = tostring(slot.value)
|
||||||
|
end
|
||||||
|
add(string.format("| `%s` | %s | %s | %s | %s |",
|
||||||
|
written_name_for(k, a),
|
||||||
|
aliases_for_key(k, a, view),
|
||||||
|
physical_for_key(k, a, view),
|
||||||
|
lattice,
|
||||||
|
last_relation_for(k, a)))
|
||||||
|
end
|
||||||
|
add("")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if not wrote then add("_(none)_"); add("") end
|
||||||
|
end
|
||||||
|
|
||||||
|
local SECTION_RENDERERS = {
|
||||||
|
{ header = "## Declarations", render = render_section_declarations },
|
||||||
|
{ header = "## Components", render = render_section_components },
|
||||||
|
{ header = "## RegUse schemas", render = render_section_reguse },
|
||||||
|
{ header = "## Annotations", render = render_section_annotations },
|
||||||
|
{ header = "## Component annotations", render = render_section_component_annotations },
|
||||||
|
{ header = "## Binds_* structs", render = render_section_binds },
|
||||||
|
{ header = "## Phases / views / ctx", render = render_section_phases },
|
||||||
|
{ header = "## Register aliases", render = render_section_aliases },
|
||||||
|
{ header = "## Auto-reg", render = render_section_autoreg },
|
||||||
|
{ header = "## Collisions", render = render_section_collisions },
|
||||||
|
{ header = "## Findings", render = render_section_findings },
|
||||||
|
{ header = "## Relations", render = render_section_relations },
|
||||||
|
{ header = "## GPR model", render = render_section_forward },
|
||||||
|
}
|
||||||
|
|
||||||
|
--- Render the consolidated per-module markdown (`build/<module>.atom_meta_report.md`).
|
||||||
|
--- One ModuleView from the corpus; SECTION_RENDERERS walks it.
|
||||||
|
--- @param view table
|
||||||
|
--- @return string
|
||||||
|
local function render_module_meta_report(view)
|
||||||
|
local dir_basename = source_basename(view.dir)
|
||||||
local lines = {
|
local lines = {
|
||||||
"# " .. dir_basename .. " — atom meta report",
|
"# " .. dir_basename .. " — atom meta report",
|
||||||
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
||||||
@@ -264,199 +786,41 @@ local function render_module_meta_report(dir, dir_sources, annot_results, sa_res
|
|||||||
}
|
}
|
||||||
local function add(s) lines[#lines + 1] = s end
|
local function add(s) lines[#lines + 1] = s end
|
||||||
|
|
||||||
-- Module summary table.
|
local kinds = count_kinds(view.decls)
|
||||||
local n_atoms = 0
|
local n_annot, n_binds, n_macros = 0, 0, 0
|
||||||
local n_annot = 0
|
for _, src in ipairs(view.sources) do
|
||||||
local n_binds = 0
|
n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
|
||||||
local n_macros = 0
|
n_binds = n_binds + #((src.scan and src.scan.binds) or {})
|
||||||
local n_bare, n_proc = 0, 0
|
n_macros = n_macros + #((src.scan and src.scan.macros) or {})
|
||||||
for _, r in ipairs(annot_results) do
|
|
||||||
n_atoms = n_atoms + #r.atoms
|
|
||||||
n_annot = n_annot + #r.annots
|
|
||||||
n_binds = n_binds + #r.binds
|
|
||||||
n_macros = n_macros + #r.macros
|
|
||||||
end
|
end
|
||||||
for _, a in ipairs(sa_results.atoms or {}) do
|
local n_err, n_warn, n_info = 0, 0, 0
|
||||||
if a.kind == "comp_bare" then n_bare = n_bare + 1
|
for _, f in ipairs(view.findings or {}) do
|
||||||
elseif a.kind == "comp_proc" then n_proc = n_proc + 1
|
if f.kind == "error" then n_err = n_err + 1
|
||||||
|
elseif f.kind == "warning" then n_warn = n_warn + 1
|
||||||
|
else n_info = n_info + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
add("## Module summary"); add("")
|
add("## Module summary"); add("")
|
||||||
add("| metric | value |"); add("|--------|-------|")
|
add("| metric | value |"); add("|--------|-------|")
|
||||||
add(string.format("| sources | %d |", #dir_sources))
|
add(string.format("| sources | %d |", #view.sources))
|
||||||
add(string.format("| atoms | %d (atoms: %d, comp_bare: %d, comp_proc: %d) |",
|
add(string.format("| decls | %d (atom: %d, atom_proc: %d, comp_bare: %d, comp_proc: %d) |",
|
||||||
#(sa_results.atoms or {}),
|
#view.decls, kinds.atom, kinds.atom_proc, kinds.comp_bare, kinds.comp_proc))
|
||||||
#(sa_results.atoms or {}) - n_bare - n_proc, n_bare, n_proc))
|
|
||||||
add(string.format("| annotations | %d |", n_annot))
|
add(string.format("| annotations | %d |", n_annot))
|
||||||
add(string.format("| binds structs | %d |", n_binds))
|
add(string.format("| binds structs | %d |", n_binds))
|
||||||
add(string.format("| macro decls | %d |", n_macros))
|
add(string.format("| macro decls | %d |", n_macros))
|
||||||
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
|
add(string.format("| findings | %d (errors: %d, warnings: %d, info: %d) |",
|
||||||
#(sa_results.findings or {}),
|
#(view.findings or {}), n_err, n_warn, n_info))
|
||||||
#(sa_results.errors or {}),
|
|
||||||
#(sa_results.warnings or {}),
|
|
||||||
#(sa_results.info or {})))
|
|
||||||
add("")
|
add("")
|
||||||
|
|
||||||
-- Sources
|
|
||||||
add("## Sources"); add("")
|
add("## Sources"); add("")
|
||||||
for _, s in ipairs(dir_sources) do add("- `" .. s.path .. "`") end
|
for _, s in ipairs(view.sources) do add("- `" .. s.path .. "`") end
|
||||||
add("")
|
add("")
|
||||||
|
|
||||||
-- Atoms (annotation)
|
for _, row in ipairs(SECTION_RENDERERS) do
|
||||||
add("## Atoms"); add("")
|
add(row.header); add("")
|
||||||
add("| kind | name | source | line |"); add("|------|------|--------|------|")
|
row.render(add, view)
|
||||||
for _, r in ipairs(annot_results) do
|
|
||||||
local src_name = source_basename(r.source)
|
|
||||||
for _, a in ipairs(r.atoms) do
|
|
||||||
add(string.format("| atom | %s | %s | %d |", a.name, src_name, a.line))
|
|
||||||
end
|
end
|
||||||
end
|
|
||||||
add("")
|
|
||||||
|
|
||||||
-- Annotations
|
|
||||||
add("## Annotations"); add("")
|
|
||||||
if #annot_results == 0 then
|
|
||||||
add("_(none)_")
|
|
||||||
else
|
|
||||||
add("| source | line | name | binds | reads | writes |")
|
|
||||||
add("|--------|------|------|-------|-------|--------|")
|
|
||||||
for _, r in ipairs(annot_results) do
|
|
||||||
local src_name = source_basename(r.source)
|
|
||||||
for _, a in ipairs(r.annots) do
|
|
||||||
local binds = a.binds or "—"
|
|
||||||
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or "—"
|
|
||||||
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or "—"
|
|
||||||
add(string.format("| %s | %d | %s | %s | %s | %s |"
|
|
||||||
, src_name, a.line, a.name, binds, reads, writes))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
|
|
||||||
-- Binds_* structs
|
|
||||||
add("## Binds_* structs"); add("")
|
|
||||||
if #annot_results == 0 then
|
|
||||||
add("_(none)_")
|
|
||||||
else
|
|
||||||
for _, r in ipairs(annot_results) do
|
|
||||||
local src_name = source_basename(r.source)
|
|
||||||
for _, b in ipairs(r.binds) do
|
|
||||||
add(string.format("### %s (%s:%d, %d bytes)",
|
|
||||||
b.name, src_name, b.line, b.bytes))
|
|
||||||
for _, f in ipairs(b.fields) do
|
|
||||||
add(string.format("- `+%d %s`", f.offset, f.name))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Macro decls
|
|
||||||
add("## Macro word-count declarations"); add("")
|
|
||||||
if #annot_results == 0 then
|
|
||||||
add("_(none)_")
|
|
||||||
else
|
|
||||||
add("| source | line | macro declaration |")
|
|
||||||
add("|--------|------|-------------------|")
|
|
||||||
for _, r in ipairs(annot_results) do
|
|
||||||
local src_name = source_basename(r.source)
|
|
||||||
for _, m in ipairs(r.macros) do
|
|
||||||
add(string.format("| %s | %d | %s |",
|
|
||||||
src_name, m.line, m.name))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
|
|
||||||
-- Findings by atom (static-analysis)
|
|
||||||
add("## Static analysis — findings by atom"); add("")
|
|
||||||
local by_atom = {}
|
|
||||||
for _, f in ipairs(sa_results.findings or {}) do
|
|
||||||
by_atom[f.atom] = by_atom[f.atom] or {}
|
|
||||||
by_atom[f.atom][#by_atom[f.atom] + 1] = f
|
|
||||||
end
|
|
||||||
if next(by_atom) == nil then
|
|
||||||
add("_(no findings)_")
|
|
||||||
else
|
|
||||||
for _, a in ipairs(sa_results.atoms or {}) do
|
|
||||||
local fs = by_atom[a.name]
|
|
||||||
if fs then
|
|
||||||
add(string.format("### %s", a.name))
|
|
||||||
for _, f in ipairs(fs) do
|
|
||||||
add(string.format("- `[%s] %s`", f.check, f.msg))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- Errors / Warnings / Info
|
|
||||||
local function add_findings(label, entries)
|
|
||||||
add(string.format("## %s", label))
|
|
||||||
if #entries == 0 then
|
|
||||||
add("_(none)_")
|
|
||||||
else
|
|
||||||
for _, e in ipairs(entries) do
|
|
||||||
add(string.format("- line %d %s", e.line, e.msg))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
end
|
|
||||||
add_findings("Errors", sa_results.errors or {})
|
|
||||||
add_findings("Warnings", sa_results.warnings or {})
|
|
||||||
add_findings("Info", sa_results.info or {})
|
|
||||||
|
|
||||||
-- Per-atom cycle counts (path-aware)
|
|
||||||
add("## Per-atom cycle counts (path-aware, best case, no stalls)"); add("")
|
|
||||||
add("| atom | source | min | max | branches | paths | notes |")
|
|
||||||
add("|------|--------|-----|-----|----------|-------|-------|")
|
|
||||||
local sorted = {}
|
|
||||||
for _, a in ipairs(sa_results.atoms or {}) do sorted[#sorted + 1] = a end
|
|
||||||
table.sort(sorted, function(x, y)
|
|
||||||
return ((x.paths or {}).cycles_max or 0) > ((y.paths or {}).cycles_max or 0)
|
|
||||||
end)
|
|
||||||
for _, a in ipairs(sorted) do
|
|
||||||
local p = a.paths or {}
|
|
||||||
local src_name = a.source_path and source_basename(a.source_path) or ""
|
|
||||||
local notes = ""
|
|
||||||
if p.has_loops then notes = notes .. " [loop!]" end
|
|
||||||
if p.unknown_macros and #p.unknown_macros > 0 then
|
|
||||||
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
|
|
||||||
end
|
|
||||||
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
|
|
||||||
a.name, src_name,
|
|
||||||
p.cycles_min or 0, p.cycles_max or 0,
|
|
||||||
p.branches or 0, p.paths or 0, notes))
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
|
|
||||||
-- Per-source scan summary
|
|
||||||
add("## Per-source scan summary"); add("")
|
|
||||||
for _, src in ipairs(dir_sources) do
|
|
||||||
local src_atoms = {}
|
|
||||||
for _, a in ipairs(sa_results.atoms or {}) do
|
|
||||||
if a.source_path == src.path then src_atoms[#src_atoms + 1] = a end
|
|
||||||
end
|
|
||||||
if #src_atoms > 0 then
|
|
||||||
local mn, mx = math.huge, -1
|
|
||||||
for _, a in ipairs(src_atoms) do
|
|
||||||
local p = a.paths or {}
|
|
||||||
if (p.cycles_min or 0) < mn then mn = p.cycles_min or 0 end
|
|
||||||
if (p.cycles_max or 0) > mx then mx = p.cycles_max or 0 end
|
|
||||||
end
|
|
||||||
local path_str
|
|
||||||
if mx > 0 then
|
|
||||||
path_str = string.format(" cycles=%d..%d", mn, mx)
|
|
||||||
else
|
|
||||||
path_str = string.format(" %d cycles", mn)
|
|
||||||
end
|
|
||||||
add(string.format("- `%s` — %d atom%s%s",
|
|
||||||
src.basename, #src_atoms,
|
|
||||||
#src_atoms == 1 and "" or "s", path_str))
|
|
||||||
end
|
|
||||||
end
|
|
||||||
add("")
|
|
||||||
|
|
||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
@@ -474,19 +838,8 @@ local REPORT_RENDERERS = {
|
|||||||
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
|
basename = function(dir_basename) return dir_basename .. ".atom_meta_report" end,
|
||||||
once = false,
|
once = false,
|
||||||
gather = function(ctx, dir, dir_sources)
|
gather = function(ctx, dir, dir_sources)
|
||||||
-- Annotations: re-run `annotation.validate()` per source (the existing pattern).
|
local corpus = ctx.shared.corpus
|
||||||
local annot_results = {}
|
return render_module_meta_report(build_module_view(dir, dir_sources, corpus))
|
||||||
for _, src in ipairs(dir_sources) do
|
|
||||||
if src.scan then
|
|
||||||
local r = annotation.validate(ctx, src, nil)
|
|
||||||
r.source = src.path
|
|
||||||
annot_results[#annot_results + 1] = r
|
|
||||||
end
|
|
||||||
end
|
|
||||||
-- Static-analysis: read stashed projection (no re-validate).
|
|
||||||
local dir_basename = dir:match("([^/\\]+)$") or dir
|
|
||||||
local sa_results = (ctx.shared.corpus.static_analysis_results or {})[dir_basename] or {}
|
|
||||||
return render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
|
||||||
end,
|
end,
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
@@ -554,32 +907,30 @@ function M.run(ctx)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
-- For the summary, compute per-module totals once (re-validating annotations per source — same pattern as the meta_report renderer).
|
local view = build_module_view(dir, dir_sources, corpus)
|
||||||
local annot_results = {}
|
|
||||||
for _, src in ipairs(dir_sources) do
|
|
||||||
if src.scan then
|
|
||||||
local r = annotation.validate(ctx, src, nil)
|
|
||||||
r.source = src.path
|
|
||||||
annot_results[#annot_results + 1] = r
|
|
||||||
end
|
|
||||||
end
|
|
||||||
local n_annot, n_binds, n_macros = 0, 0, 0
|
local n_annot, n_binds, n_macros = 0, 0, 0
|
||||||
for _, r in ipairs(annot_results) do
|
for _, src in ipairs(dir_sources) do
|
||||||
n_annot = n_annot + #r.annots
|
n_annot = n_annot + #((src.scan and src.scan.atom_infos) or {})
|
||||||
n_binds = n_binds + #r.binds
|
n_binds = n_binds + #((src.scan and src.scan.binds) or {})
|
||||||
n_macros = n_macros + #r.macros
|
n_macros = n_macros + #((src.scan and src.scan.macros) or {})
|
||||||
|
end
|
||||||
|
local n_err, n_warn, n_info = 0, 0, 0
|
||||||
|
for _, f in ipairs(view.findings or {}) do
|
||||||
|
if f.kind == "error" then n_err = n_err + 1
|
||||||
|
elseif f.kind == "warning" then n_warn = n_warn + 1
|
||||||
|
else n_info = n_info + 1
|
||||||
|
end
|
||||||
end
|
end
|
||||||
local sa_results = (corpus.static_analysis_results or {})[dir_basename] or {}
|
|
||||||
all_modules[#all_modules + 1] = {
|
all_modules[#all_modules + 1] = {
|
||||||
module = dir_basename,
|
module = dir_basename,
|
||||||
atoms = #(sa_results.atoms or {}),
|
atoms = #view.decls,
|
||||||
annots = n_annot,
|
annots = n_annot,
|
||||||
binds = n_binds,
|
binds = n_binds,
|
||||||
macros = n_macros,
|
macros = n_macros,
|
||||||
findings = #(sa_results.findings or {}),
|
findings = #(view.findings or {}),
|
||||||
errors = #(sa_results.errors or {}),
|
errors = n_err,
|
||||||
warnings = #(sa_results.warnings or {}),
|
warnings = n_warn,
|
||||||
info = #(sa_results.info or {}),
|
info = n_info,
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
+691
-141
@@ -6,6 +6,7 @@
|
|||||||
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
|
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
|
||||||
--- MipsAtomComp_ (kind = "comp_bare")
|
--- MipsAtomComp_ (kind = "comp_bare")
|
||||||
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
|
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
|
||||||
|
--- MipsAtomComp_ProcMap_ (kind = "comp_proc", body is the one command)
|
||||||
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
|
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
|
||||||
--- MipsCode code_<name> (kind = "raw_atom", offsets pass only)
|
--- MipsCode code_<name> (kind = "raw_atom", offsets pass only)
|
||||||
--- typedef Struct_(Binds_X) { fields }
|
--- typedef Struct_(Binds_X) { fields }
|
||||||
@@ -415,31 +416,51 @@ local function walk_body_fields(body, build_field)
|
|||||||
return fields
|
return fields
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Parse the `<type> <field>;` declarations from a Struct_ body.
|
-- Parse the `<type> <field>[, <field>...];` declarations from a Struct_ body.
|
||||||
|
-- After the type and `*` chain, keep reading `, ident` until `;`.
|
||||||
|
-- Same type, same pointer depth for every name on that list.
|
||||||
-- Returns the raw fields array with `{name, type_name, pointer_depth}` only (NO offset / byte_size).
|
-- Returns the raw fields array with `{name, type_name, pointer_depth}` only (NO offset / byte_size).
|
||||||
-- The propagation pass `resolve_struct_field_sizes` walks each struct's fields AFTER type resolution and populates offset + byte_size in place.
|
-- The propagation pass `resolve_struct_field_sizes` walks each struct's fields AFTER type resolution and populates offset + byte_size in place.
|
||||||
-- Returns (fields). The aggregate byte_count is computed in the propagation pass (it depends on whether every field's type resolved).
|
|
||||||
local function parse_struct_body_fields(body)
|
local function parse_struct_body_fields(body)
|
||||||
return walk_body_fields(body, function(type_name, type_end, after_type)
|
local fields = {}
|
||||||
-- Parse the trailing `*` chain to derive pointer_depth.
|
local body_pos = 1
|
||||||
local depth, cursor = 0, after_type
|
local body_len = #body
|
||||||
while cursor <= #body and body:sub(cursor, cursor) == "*" do
|
while body_pos <= body_len do
|
||||||
|
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
|
||||||
|
if body_pos > body_len then break end
|
||||||
|
local type_name, type_end = duffle.read_ident(body, body_pos)
|
||||||
|
if not type_name then
|
||||||
|
body_pos = body_pos + 1
|
||||||
|
else
|
||||||
|
local depth, cursor = 0, duffle.skip_ws_and_cmt(body, type_end)
|
||||||
|
while cursor <= body_len and body:sub(cursor, cursor) == "*" do
|
||||||
depth = depth + 1
|
depth = depth + 1
|
||||||
cursor = cursor + 1
|
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
|
||||||
cursor = duffle.skip_ws_and_cmt(body, cursor)
|
|
||||||
end
|
end
|
||||||
-- Read the field ident immediately after the type chain.
|
while cursor <= body_len do
|
||||||
local field_ident, field_end = duffle.read_ident(body, cursor)
|
local field_ident, field_end = duffle.read_ident(body, cursor)
|
||||||
if not field_ident then return nil, type_end + 1 end
|
if not field_ident then break end
|
||||||
return {
|
fields[#fields + 1] = {
|
||||||
name = field_ident,
|
name = field_ident,
|
||||||
type_name = type_name,
|
type_name = type_name,
|
||||||
pointer_depth = depth,
|
pointer_depth = depth,
|
||||||
-- offset + byte_size filled by resolve_struct_field_sizes
|
|
||||||
offset = nil,
|
offset = nil,
|
||||||
byte_size = nil,
|
byte_size = nil,
|
||||||
}, field_end
|
}
|
||||||
end)
|
cursor = duffle.skip_ws_and_cmt(body, field_end)
|
||||||
|
if body:sub(cursor, cursor) == "," then
|
||||||
|
cursor = duffle.skip_ws_and_cmt(body, cursor + 1)
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if cursor <= body_len and body:sub(cursor, cursor) == ";" then
|
||||||
|
cursor = cursor + 1
|
||||||
|
end
|
||||||
|
body_pos = cursor
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return fields
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries.
|
-- Parse the `Enum_(<underlying>, <name>) { <body> }` body for entries.
|
||||||
@@ -1251,31 +1272,20 @@ local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
|
|||||||
return after_paren
|
return after_paren
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
|
--- Lookahead for `atom_info(...)` after a declaration's closing paren.
|
||||||
--- @param source string
|
--- Records into `dest` (atom_infos or component_atom_infos). Returns the position after the info, or after_paren if none.
|
||||||
--- @param pos integer
|
local function parse_atom_info_after_decl(source, after_paren, raw_name, line_of, out, dest)
|
||||||
--- @param ident_end integer
|
|
||||||
--- @param line_of fun(pos: integer): integer
|
|
||||||
--- @param out SourceScan
|
|
||||||
--- @return integer
|
|
||||||
local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
|
||||||
if not inner then return after_paren end
|
|
||||||
|
|
||||||
local raw_name = duffle.read_ident(inner, 1)
|
|
||||||
|
|
||||||
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
|
|
||||||
local brace_search_pos = after_paren
|
|
||||||
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
|
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
|
||||||
local look_ident, look_end = duffle.read_ident(source, lookahead)
|
local look_ident, look_end = duffle.read_ident(source, lookahead)
|
||||||
if look_ident == "atom_info" then
|
if look_ident ~= "atom_info" then return after_paren end
|
||||||
local info_open = duffle.skip_ws_and_cmt(source, look_end)
|
local info_open = duffle.skip_ws_and_cmt(source, look_end)
|
||||||
if source:sub(info_open, info_open) == "(" then
|
if source:sub(info_open, info_open) ~= "(" then return after_paren end
|
||||||
local info_inner, info_after = duffle.read_parens(source, info_open)
|
local info_inner, info_after = duffle.read_parens(source, info_open)
|
||||||
-- info_line feeds the per-atom reg_type_overrides table.
|
if not info_inner then return after_paren end
|
||||||
local info_line = line_of(info_open)
|
local info_line = line_of(info_open)
|
||||||
local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line)
|
local ai_binds, ai_reads, ai_writes, ai_view, ai_overrides, ai_ctx, ai_phase = scan_atom_info_subcalls(info_inner, info_line)
|
||||||
out.atom_infos[#out.atom_infos + 1] = {
|
dest = dest or out.atom_infos
|
||||||
|
dest[#dest + 1] = {
|
||||||
atom_name = raw_name or "?", binds = ai_binds,
|
atom_name = raw_name or "?", binds = ai_binds,
|
||||||
reads = ai_reads or {}, writes = ai_writes or {},
|
reads = ai_reads or {}, writes = ai_writes or {},
|
||||||
view = ai_view,
|
view = ai_view,
|
||||||
@@ -1292,11 +1302,9 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
|||||||
info_line = line_of(lookahead),
|
info_line = line_of(lookahead),
|
||||||
}
|
}
|
||||||
elseif raw_name and ai_overrides then
|
elseif raw_name and ai_overrides then
|
||||||
-- Record per-atom overrides even without atom_view.
|
|
||||||
out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) }
|
out.atom_views[raw_name] = out.atom_views[raw_name] or { atom_name = raw_name, binds_name = nil, reg_type_overrides = nil, info_line = line_of(lookahead) }
|
||||||
out.atom_views[raw_name].reg_type_overrides = ai_overrides
|
out.atom_views[raw_name].reg_type_overrides = ai_overrides
|
||||||
end
|
end
|
||||||
-- Project the per-atom atom_ctx / atom_phase declarations onto the global phase index.
|
|
||||||
if raw_name then
|
if raw_name then
|
||||||
if ai_ctx then
|
if ai_ctx then
|
||||||
out.atom_ctxs = out.atom_ctxs or {}
|
out.atom_ctxs = out.atom_ctxs or {}
|
||||||
@@ -1308,122 +1316,163 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
|||||||
out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name
|
out.atom_phases[ai_phase].atoms[#out.atom_phases[ai_phase].atoms + 1] = raw_name
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
brace_search_pos = info_after
|
return info_after
|
||||||
end
|
|
||||||
end
|
end
|
||||||
|
|
||||||
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
|
local DECL_FORMS = {
|
||||||
if not body then return after_brace end
|
MipsAtom_ = {
|
||||||
if raw_name and raw_name ~= "" then
|
kind = "atom", name = "paren_ident", body = "braces_after",
|
||||||
register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source)
|
info_dest = "atom_infos", strip = false,
|
||||||
end
|
},
|
||||||
|
MipsAtom_Proc_ = {
|
||||||
|
kind = "atom_proc", name = "backward_atom_proc", body = "last_brace_in_args",
|
||||||
|
info_dest = "atom_infos", strip = false, after = "reguse_hook",
|
||||||
|
},
|
||||||
|
MipsAtomComp_ = {
|
||||||
|
kind = "comp_bare", name = "paren_ident", body = "braces_after",
|
||||||
|
info_dest = "component_atom_infos", strip = "ac_",
|
||||||
|
},
|
||||||
|
MipsAtomComp_Proc_ = {
|
||||||
|
kind = "comp_proc", name = "backward_fi", body = "last_brace_in_args",
|
||||||
|
info_dest = nil, strip = "ac_",
|
||||||
|
},
|
||||||
|
MipsAtomComp_ProcMap_ = {
|
||||||
|
kind = "comp_proc", name = "backward_fi", body = "comma_arg_2",
|
||||||
|
info_dest = nil, strip = "ac_", after = "map_command_hook",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
return after_brace
|
local function last_brace_body(inner, open_paren)
|
||||||
end
|
|
||||||
|
|
||||||
--- Parse: `MipsAtomComp_(<name>) { <body> }`
|
|
||||||
--- @param source string
|
|
||||||
--- @param pos integer
|
|
||||||
--- @param ident_end integer
|
|
||||||
--- @param line_of fun(pos: integer): integer
|
|
||||||
--- @param out SourceScan
|
|
||||||
--- @return integer
|
|
||||||
local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
|
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
|
||||||
if not inner then return after_paren end
|
|
||||||
|
|
||||||
local raw_name = duffle.read_ident(inner, 1)
|
|
||||||
if not raw_name then return open_paren + 1 end
|
|
||||||
|
|
||||||
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
|
|
||||||
if not body then return after_brace end
|
|
||||||
local name = strip_ac_prefix(raw_name)
|
|
||||||
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
|
||||||
|
|
||||||
return after_brace
|
|
||||||
end
|
|
||||||
|
|
||||||
--- Parse: `MipsAtomComp_Proc_(<name>, { <body> })` — body is inside the LAST `{` in args.
|
|
||||||
--- @param source string
|
|
||||||
--- @param pos integer
|
|
||||||
--- @param ident_end integer
|
|
||||||
--- @param line_of fun(pos: integer): integer
|
|
||||||
--- @param out SourceScan
|
|
||||||
--- @return integer
|
|
||||||
local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
|
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
|
||||||
if not inner then return after_paren end
|
|
||||||
|
|
||||||
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
|
|
||||||
local last_brace_pos = nil
|
local last_brace_pos = nil
|
||||||
for search_pos = #inner, 1, -1 do
|
for search_pos = #inner, 1, -1 do
|
||||||
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
|
if inner:sub(search_pos, search_pos) == "{" then
|
||||||
|
last_brace_pos = search_pos
|
||||||
|
break
|
||||||
end
|
end
|
||||||
if not last_brace_pos then return after_paren end
|
end
|
||||||
|
if not last_brace_pos then return nil end
|
||||||
-- Use duffle.read_braces to find the matching close brace.
|
|
||||||
-- Uses `read_balanced` for delimiter-depth tracking.
|
|
||||||
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
|
|
||||||
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
||||||
if close_pos > #inner + 1 then return after_paren end
|
if close_pos > #inner + 1 then return nil end
|
||||||
|
return body, open_paren + 2 + last_brace_pos
|
||||||
|
end
|
||||||
|
|
||||||
-- The component name is derived from the preceding function declaration
|
local function reguse_hook(source, pos, line_of, out, extras)
|
||||||
-- (`FI_ Slice_MipsCode ac_X(...)`), not from the first macro arg (which
|
local entry = out.atoms[#out.atoms]
|
||||||
-- is now `ab`). The backward walk finds the function decl before open_paren.
|
if not entry then return end
|
||||||
local raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
|
local reg_use_schema_name, reg_use_param_name
|
||||||
|
if extras.args_inner then
|
||||||
|
local arg_tokens = duffle.split_top_level_commas(extras.args_inner)
|
||||||
|
for _, tok in ipairs(arg_tokens) do
|
||||||
|
local trimmed = duffle.trim(tok)
|
||||||
|
local schema_suffix, param = trimmed:match("RegUse_([%w_]+)%s+([%w_]+)$")
|
||||||
|
if schema_suffix then
|
||||||
|
if reg_use_schema_name then
|
||||||
|
out.reg_use_errors[#out.reg_use_errors + 1] = {
|
||||||
|
kind = "reguse_multiple_params",
|
||||||
|
schema_name = "RegUse_" .. schema_suffix,
|
||||||
|
source_line = line_of(pos),
|
||||||
|
}
|
||||||
|
else
|
||||||
|
reg_use_schema_name = "RegUse_" .. schema_suffix
|
||||||
|
reg_use_param_name = param
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
entry.reg_use_schema_name = reg_use_schema_name
|
||||||
|
entry.reg_use_param_name = reg_use_param_name
|
||||||
|
if reg_use_schema_name and extras.func_ident then
|
||||||
|
local expected = "RegUse_" .. extras.func_ident
|
||||||
|
if reg_use_schema_name ~= expected then
|
||||||
|
out.reg_use_errors[#out.reg_use_errors + 1] = {
|
||||||
|
kind = "reguse_name_mismatch",
|
||||||
|
schema_name = reg_use_schema_name,
|
||||||
|
func_ident = extras.func_ident,
|
||||||
|
source_line = line_of(pos),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local function parse_decl_form(source, pos, ident_end, line_of, out)
|
||||||
|
local ident = duffle.read_ident(source, pos)
|
||||||
|
local form = ident and DECL_FORMS[ident]
|
||||||
|
if not form then return ident_end end
|
||||||
|
|
||||||
|
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||||
|
if not inner then return after_paren end
|
||||||
|
|
||||||
|
local extras = {}
|
||||||
|
local raw_name
|
||||||
|
if form.name == "paren_ident" then
|
||||||
|
raw_name = duffle.read_ident(inner, 1)
|
||||||
|
if form.strip and not raw_name then return open_paren + 1 end
|
||||||
|
elseif form.name == "backward_fi" then
|
||||||
|
raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
|
||||||
|
elseif form.name == "backward_atom_proc" then
|
||||||
|
raw_name, extras.args_inner, extras.func_ident, extras.after_func_paren =
|
||||||
|
duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
|
||||||
|
end
|
||||||
|
if form.name == "paren_ident" and form.strip and not raw_name then
|
||||||
|
return open_paren + 1
|
||||||
|
end
|
||||||
if not raw_name then raw_name = "?" end
|
if not raw_name then raw_name = "?" end
|
||||||
local name = strip_ac_prefix(raw_name)
|
local name = form.strip and strip_ac_prefix(raw_name) or raw_name
|
||||||
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
|
|
||||||
local body_off = open_paren + 2 + last_brace_pos
|
|
||||||
register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
|
||||||
|
|
||||||
|
if form.info_dest == "component_atom_infos" then
|
||||||
|
out.component_atom_infos = out.component_atom_infos or {}
|
||||||
|
end
|
||||||
|
local info_dest = form.info_dest and out[form.info_dest]
|
||||||
|
|
||||||
|
local body, body_off, resume
|
||||||
|
if form.body == "braces_after" then
|
||||||
|
local brace_search = after_paren
|
||||||
|
if info_dest then
|
||||||
|
brace_search = parse_atom_info_after_decl(
|
||||||
|
source, after_paren, name, line_of, out, info_dest)
|
||||||
|
end
|
||||||
|
local after_brace
|
||||||
|
body, after_brace, body_off = find_body_braces(source, brace_search, open_paren + 1)
|
||||||
|
if not body then return after_brace end
|
||||||
|
resume = after_brace
|
||||||
|
elseif form.body == "last_brace_in_args" then
|
||||||
|
body, body_off = last_brace_body(inner, open_paren)
|
||||||
|
if not body then return after_paren end
|
||||||
|
resume = after_paren
|
||||||
|
if form.info_dest then
|
||||||
|
if extras.after_func_paren then
|
||||||
|
parse_atom_info_after_decl(
|
||||||
|
source, extras.after_func_paren, name, line_of, out, out.atom_infos)
|
||||||
|
else
|
||||||
|
parse_atom_info_after_decl(source, pos, name, line_of, out, out.atom_infos)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
elseif form.body == "comma_arg_2" then
|
||||||
|
local args = duffle.split_top_level_commas(inner)
|
||||||
|
if #args < 2 then return after_paren end
|
||||||
|
body = duffle.trim(args[2])
|
||||||
|
if body == "" then return after_paren end
|
||||||
|
body_off = open_paren + 1 + (inner:find(body, 1, true) or 1) - 1
|
||||||
|
resume = after_paren
|
||||||
|
else
|
||||||
return after_paren
|
return after_paren
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
|
local skip_register = form.name == "paren_ident" and not form.strip
|
||||||
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
|
and (raw_name == "?" or raw_name == "")
|
||||||
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit
|
if not skip_register then
|
||||||
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
|
register_atom(out, form.kind, line_of(pos), name, body, body_off,
|
||||||
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
|
raw_name, pos, after_paren, source)
|
||||||
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
|
|
||||||
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
|
|
||||||
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
|
|
||||||
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
|
|
||||||
--- @param source string
|
|
||||||
--- @param pos integer
|
|
||||||
--- @param ident_end integer
|
|
||||||
--- @param line_of fun(pos: integer): integer
|
|
||||||
--- @param out SourceScan
|
|
||||||
--- @return integer
|
|
||||||
local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
|
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
|
||||||
if not inner then return after_paren end
|
|
||||||
|
|
||||||
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
|
|
||||||
local last_brace_pos = nil
|
|
||||||
for search_pos = #inner, 1, -1 do
|
|
||||||
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
|
|
||||||
end
|
end
|
||||||
if not last_brace_pos then return after_paren end
|
|
||||||
|
|
||||||
-- Use duffle.read_braces to find the matching close brace.
|
if form.after == "reguse_hook" then
|
||||||
-- Uses `read_balanced` for delimiter-depth tracking.
|
reguse_hook(source, pos, line_of, out, extras)
|
||||||
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
|
elseif form.after == "map_command_hook" then
|
||||||
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
local entry = out.atoms[#out.atoms]
|
||||||
if close_pos > #inner + 1 then return after_paren end
|
if entry then entry.map_command = body end
|
||||||
|
end
|
||||||
|
|
||||||
-- The atom name is derived from the preceding function declaration
|
return resume
|
||||||
-- (`internal MipsAtom* X_proc(...)`), not from the first macro arg (which
|
|
||||||
-- is now `aa`). The backward walk finds the function decl before open_paren
|
|
||||||
-- and strips the `_proc` suffix.
|
|
||||||
local raw_name = duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
|
|
||||||
if not raw_name then raw_name = "?" end
|
|
||||||
local name = strip_ac_prefix(raw_name)
|
|
||||||
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
|
|
||||||
local body_off = open_paren + 2 + last_brace_pos
|
|
||||||
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
|
||||||
|
|
||||||
return after_paren
|
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
|
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
|
||||||
@@ -1530,6 +1579,236 @@ local function register_typedef_alias(underlying, name, pos, line_of, out)
|
|||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
|
local parse_reg_use_schema_body
|
||||||
|
|
||||||
|
local function fields_for_reg_type(type_name, type_registry)
|
||||||
|
local reg_name = "Reg_" .. type_name
|
||||||
|
local entry = type_registry and type_registry[reg_name]
|
||||||
|
if entry and entry.fields and #entry.fields > 0 then
|
||||||
|
local names = {}
|
||||||
|
for _, field in ipairs(entry.fields) do
|
||||||
|
if field.name then names[#names + 1] = field.name end
|
||||||
|
end
|
||||||
|
if #names > 0 then return names end
|
||||||
|
end
|
||||||
|
if entry and entry.body and parse_reg_use_schema_body then
|
||||||
|
local schema = parse_reg_use_schema_body(entry.body, type_registry)
|
||||||
|
if schema and schema.slots then
|
||||||
|
local names = {}
|
||||||
|
for _, slot in ipairs(schema.slots) do
|
||||||
|
if slot.name then names[#names + 1] = slot.name end
|
||||||
|
end
|
||||||
|
if #names > 0 then return names end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
parse_reg_use_schema_body = function(body, type_registry, opts)
|
||||||
|
opts = opts or {}
|
||||||
|
local require_types = opts.require_types == true
|
||||||
|
local pending = false
|
||||||
|
local slots = {}
|
||||||
|
local alias_to_slot = {}
|
||||||
|
local slot_names = {}
|
||||||
|
local errors = {}
|
||||||
|
|
||||||
|
local function add_alias(path, slot)
|
||||||
|
if alias_to_slot[path] then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_duplicate_alias", path = path }
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
alias_to_slot[path] = slot
|
||||||
|
return true
|
||||||
|
end
|
||||||
|
|
||||||
|
local function add_slot(name, aliases, readonly)
|
||||||
|
if slot_names[name] then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_duplicate_slot", name = name }
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
slot_names[name] = true
|
||||||
|
local slot = { name = name, aliases = aliases, readonly = readonly == true }
|
||||||
|
slots[#slots + 1] = slot
|
||||||
|
return slot
|
||||||
|
end
|
||||||
|
|
||||||
|
local function parse_reg_names(text, pos)
|
||||||
|
local names = {}
|
||||||
|
while pos <= #text do
|
||||||
|
pos = duffle.skip_ws_and_cmt(text, pos)
|
||||||
|
local name, name_end = duffle.read_ident(text, pos)
|
||||||
|
if not name then return nil, pos end
|
||||||
|
names[#names + 1] = name
|
||||||
|
pos = duffle.skip_ws_and_cmt(text, name_end)
|
||||||
|
if text:sub(pos, pos) == "," then
|
||||||
|
pos = pos + 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if text:sub(pos, pos) == ";" then pos = pos + 1 end
|
||||||
|
return names, pos
|
||||||
|
end
|
||||||
|
|
||||||
|
local pos = 1
|
||||||
|
while pos <= #body do
|
||||||
|
pos = duffle.skip_ws_and_cmt(body, pos)
|
||||||
|
if pos > #body then break end
|
||||||
|
local first, first_end = duffle.read_ident(body, pos)
|
||||||
|
if not first then
|
||||||
|
pos = pos + 1
|
||||||
|
goto continue
|
||||||
|
end
|
||||||
|
local after = duffle.skip_ws_and_cmt(body, first_end)
|
||||||
|
if first == "const" then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_const_reg_spelling" }
|
||||||
|
return nil, errors
|
||||||
|
elseif first == "union" then
|
||||||
|
if body:sub(after, after) ~= "{" then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local inner, after_braces = duffle.read_braces(body, after)
|
||||||
|
if not inner then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local members = {}
|
||||||
|
local union_readonly = nil
|
||||||
|
local inner_pos = 1
|
||||||
|
while inner_pos <= #inner do
|
||||||
|
inner_pos = duffle.skip_ws_and_cmt(inner, inner_pos)
|
||||||
|
if inner_pos > #inner then break end
|
||||||
|
local m_type, m_type_end = duffle.read_ident(inner, inner_pos)
|
||||||
|
if not m_type then
|
||||||
|
inner_pos = inner_pos + 1
|
||||||
|
goto continue_inner
|
||||||
|
end
|
||||||
|
if m_type == "const" then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_const_reg_spelling" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
if m_type ~= "Reg" then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local m_after = duffle.skip_ws_and_cmt(inner, m_type_end)
|
||||||
|
local m_readonly = false
|
||||||
|
local maybe_const, maybe_end = duffle.read_ident(inner, m_after)
|
||||||
|
if maybe_const == "const" then
|
||||||
|
m_readonly = true
|
||||||
|
m_after = duffle.skip_ws_and_cmt(inner, maybe_end)
|
||||||
|
end
|
||||||
|
if union_readonly == nil then
|
||||||
|
union_readonly = m_readonly
|
||||||
|
elseif union_readonly ~= m_readonly then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_mixed_const" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local names, new_inner = parse_reg_names(inner, m_after)
|
||||||
|
if not names or #names == 0 then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
for _, n in ipairs(names) do members[#members + 1] = n end
|
||||||
|
inner_pos = new_inner
|
||||||
|
::continue_inner::
|
||||||
|
end
|
||||||
|
if #members == 0 then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local after_close = duffle.skip_ws_and_cmt(body, after_braces)
|
||||||
|
local inst_name, inst_end = duffle.read_ident(body, after_close)
|
||||||
|
local aliases = {}
|
||||||
|
local slot_name
|
||||||
|
if inst_name then
|
||||||
|
slot_name = inst_name
|
||||||
|
for _, m in ipairs(members) do
|
||||||
|
local path = inst_name .. "." .. m
|
||||||
|
if not add_alias(path, slot_name) then return nil, errors end
|
||||||
|
aliases[#aliases + 1] = path
|
||||||
|
end
|
||||||
|
after_close = inst_end
|
||||||
|
else
|
||||||
|
slot_name = members[1]
|
||||||
|
for _, m in ipairs(members) do
|
||||||
|
if not add_alias(m, slot_name) then return nil, errors end
|
||||||
|
aliases[#aliases + 1] = m
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if not add_slot(slot_name, aliases, union_readonly) then return nil, errors end
|
||||||
|
after_close = duffle.skip_ws_and_cmt(body, after_close)
|
||||||
|
if body:sub(after_close, after_close) == ";" then after_close = after_close + 1 end
|
||||||
|
pos = after_close
|
||||||
|
elseif first == "Reg" or first == "Reg_" then
|
||||||
|
local typed_fields = nil
|
||||||
|
if first == "Reg_" then
|
||||||
|
if body:sub(after, after) ~= "(" then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local type_inner, after_paren = duffle.read_parens(body, after)
|
||||||
|
if not type_inner then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
local type_ident = duffle.trim(type_inner)
|
||||||
|
typed_fields = fields_for_reg_type(type_ident, type_registry)
|
||||||
|
if not typed_fields then
|
||||||
|
if require_types then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_unknown_reg_type", type_name = type_ident }
|
||||||
|
else
|
||||||
|
pending = true
|
||||||
|
end
|
||||||
|
end
|
||||||
|
after = duffle.skip_ws_and_cmt(body, after_paren)
|
||||||
|
end
|
||||||
|
local readonly = false
|
||||||
|
local maybe_const, maybe_end = duffle.read_ident(body, after)
|
||||||
|
if maybe_const == "const" then
|
||||||
|
readonly = true
|
||||||
|
after = duffle.skip_ws_and_cmt(body, maybe_end)
|
||||||
|
end
|
||||||
|
local names, new_pos = parse_reg_names(body, after)
|
||||||
|
if not names or #names == 0 then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
for _, n in ipairs(names) do
|
||||||
|
if first == "Reg_" then
|
||||||
|
if typed_fields then
|
||||||
|
for _, field in ipairs(typed_fields) do
|
||||||
|
local path = n .. "." .. field
|
||||||
|
if not add_alias(path, path) then return nil, errors end
|
||||||
|
if not add_slot(path, { path }, readonly) then return nil, errors end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
else
|
||||||
|
if not add_alias(n, n) then return nil, errors end
|
||||||
|
if not add_slot(n, { n }, readonly) then return nil, errors end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
pos = new_pos
|
||||||
|
else
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
::continue::
|
||||||
|
end
|
||||||
|
if #slots == 0 then
|
||||||
|
if pending and not require_types then
|
||||||
|
return { slots = slots, alias_to_slot = alias_to_slot, pending = true }, errors
|
||||||
|
end
|
||||||
|
if #errors == 0 then
|
||||||
|
errors[#errors + 1] = { kind = "reguse_malformed" }
|
||||||
|
end
|
||||||
|
return nil, errors
|
||||||
|
end
|
||||||
|
return { slots = slots, alias_to_slot = alias_to_slot, pending = pending }, errors
|
||||||
|
end
|
||||||
|
|
||||||
--- Parse: `typedef` declarations.
|
--- Parse: `typedef` declarations.
|
||||||
---
|
---
|
||||||
--- Recognizes four shapes:
|
--- Recognizes four shapes:
|
||||||
@@ -1563,6 +1842,21 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
|||||||
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
|
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
|
||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
register_struct_type(body, name, pos, line_of, out)
|
register_struct_type(body, name, pos, line_of, out)
|
||||||
|
if name:sub(1, 7) == "RegUse_" then
|
||||||
|
local schema, schema_errors = parse_reg_use_schema_body(body, out.type_name_registry)
|
||||||
|
if schema then
|
||||||
|
schema.name = name
|
||||||
|
schema.source_file = out._source_file
|
||||||
|
schema.source_line = line_of(pos)
|
||||||
|
out.reg_use_schemas[name] = schema
|
||||||
|
end
|
||||||
|
for _, err in ipairs(schema_errors or {}) do
|
||||||
|
err.schema_name = name
|
||||||
|
err.source_file = out._source_file
|
||||||
|
err.source_line = line_of(pos)
|
||||||
|
out.reg_use_errors[#out.reg_use_errors + 1] = err
|
||||||
|
end
|
||||||
|
end
|
||||||
attach_debug_skip_marker(out, "unrelated")
|
attach_debug_skip_marker(out, "unrelated")
|
||||||
return after_brace
|
return after_brace
|
||||||
|
|
||||||
@@ -1889,10 +2183,11 @@ end
|
|||||||
-- Adding a new construct = 1 row here + 1 parser function above.
|
-- Adding a new construct = 1 row here + 1 parser function above.
|
||||||
|
|
||||||
local DECL_PARSERS = {
|
local DECL_PARSERS = {
|
||||||
MipsAtom_ = parse_mips_atom,
|
MipsAtom_ = parse_decl_form,
|
||||||
MipsAtom_Proc_ = parse_mips_atom_proc,
|
MipsAtom_Proc_ = parse_decl_form,
|
||||||
MipsAtomComp_ = parse_mips_atom_comp,
|
MipsAtomComp_ = parse_decl_form,
|
||||||
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
|
MipsAtomComp_Proc_ = parse_decl_form,
|
||||||
|
MipsAtomComp_ProcMap_ = parse_decl_form,
|
||||||
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
|
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
|
||||||
-- identifier follows the ordinary unrelated-token path; there is no alias.
|
-- identifier follows the ordinary unrelated-token path; there is no alias.
|
||||||
atom_dbg_skip = parse_dbg_skip_marker,
|
atom_dbg_skip = parse_dbg_skip_marker,
|
||||||
@@ -1912,6 +2207,131 @@ local DECL_PARSERS = {
|
|||||||
-- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`.
|
-- Only the bare `atom_dbg_skip` marker reaches `parse_dbg_skip_marker`.
|
||||||
-- Unknown identifiers follow the same unrelated-token path as every other unsupported source token.
|
-- Unknown identifiers follow the same unrelated-token path as every other unsupported source token.
|
||||||
|
|
||||||
|
local function tape_skip_ident(ident)
|
||||||
|
return ident and (DECL_FORMS[ident] or ident == "Struct_" or ident == "Enum_")
|
||||||
|
end
|
||||||
|
|
||||||
|
local function collect_addrs_assigns_REMOVED(text)
|
||||||
|
local addrs = {}
|
||||||
|
local pos = 1
|
||||||
|
local n = #text
|
||||||
|
while pos <= n do
|
||||||
|
pos = duffle.skip_ws_and_cmt(text, pos)
|
||||||
|
if pos > n then break end
|
||||||
|
local ident, ident_end = duffle.read_ident(text, pos)
|
||||||
|
if ident == "addrs" then
|
||||||
|
local after = duffle.skip_ws_and_cmt(text, ident_end)
|
||||||
|
if text:sub(after, after) == "[" then
|
||||||
|
local inner, after_br = duffle.read_brackets(text, after)
|
||||||
|
local idx = inner and tonumber(duffle.trim(inner))
|
||||||
|
after_br = duffle.skip_ws_and_cmt(text, after_br or after)
|
||||||
|
if idx and text:sub(after_br, after_br) == "=" then
|
||||||
|
local rhs = duffle.skip_ws_and_cmt(text, after_br + 1)
|
||||||
|
local rhs_ident = duffle.read_ident(text, rhs)
|
||||||
|
if rhs_ident then addrs[idx] = rhs_ident end
|
||||||
|
pos = rhs
|
||||||
|
else
|
||||||
|
pos = after_br or (after + 1)
|
||||||
|
end
|
||||||
|
else
|
||||||
|
pos = ident_end
|
||||||
|
end
|
||||||
|
elseif ident then
|
||||||
|
pos = ident_end
|
||||||
|
else
|
||||||
|
pos = pos + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return addrs
|
||||||
|
end
|
||||||
|
|
||||||
|
local function collect_tb_emits(body, addrs)
|
||||||
|
local names = {}
|
||||||
|
local pos = 1
|
||||||
|
local n = #body
|
||||||
|
while pos <= n do
|
||||||
|
pos = duffle.skip_ws_and_cmt(body, pos)
|
||||||
|
if pos > n then break end
|
||||||
|
local ident, ident_end = duffle.read_ident(body, pos)
|
||||||
|
if ident == "tb_emit_" or ident == "tb_emit" then
|
||||||
|
local after = duffle.skip_ws_and_cmt(body, ident_end)
|
||||||
|
if body:sub(after, after) == "(" then
|
||||||
|
local inner, after_p = duffle.read_parens(body, after)
|
||||||
|
local name
|
||||||
|
if ident == "tb_emit_" then
|
||||||
|
name = duffle.trim(inner or ""):match("^([%w_]+)")
|
||||||
|
else
|
||||||
|
local args = duffle.split_top_level_commas(inner or "")
|
||||||
|
local last = duffle.trim(args[#args] or "")
|
||||||
|
local idx = last:match("^addrs%s*%[%s*(%d+)%s*%]$")
|
||||||
|
if idx then
|
||||||
|
name = addrs[tonumber(idx)]
|
||||||
|
else
|
||||||
|
name = last:match("([%w_]+)$")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if name then names[#names + 1] = name end
|
||||||
|
pos = after_p or (after + 1)
|
||||||
|
else
|
||||||
|
pos = ident_end
|
||||||
|
end
|
||||||
|
elseif ident then
|
||||||
|
pos = ident_end
|
||||||
|
else
|
||||||
|
pos = pos + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return names
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Linear appearance order of tb_emit / tb_emit_ in each C function body.
|
||||||
|
-- Commented-out emits are skipped by skip_ws_and_cmt. No C if/loop CFG.
|
||||||
|
local function scan_tape_chains(source)
|
||||||
|
local addrs = collect_addrs_assigns(source)
|
||||||
|
local chains = {}
|
||||||
|
local pos = 1
|
||||||
|
local n = #source
|
||||||
|
while pos <= n do
|
||||||
|
pos = duffle.skip_ws_and_cmt(source, pos)
|
||||||
|
if pos > n then break end
|
||||||
|
local ident, ident_end = duffle.read_ident(source, pos)
|
||||||
|
if tape_skip_ident(ident) then
|
||||||
|
local after = duffle.skip_ws_and_cmt(source, ident_end)
|
||||||
|
if source:sub(after, after) == "(" then
|
||||||
|
local _, after_p = duffle.read_parens(source, after)
|
||||||
|
after = duffle.skip_ws_and_cmt(source, after_p or after)
|
||||||
|
end
|
||||||
|
if source:sub(after, after) == "{" then
|
||||||
|
local _, after_b = duffle.read_braces(source, after)
|
||||||
|
pos = after_b or (after + 1)
|
||||||
|
else
|
||||||
|
pos = after
|
||||||
|
end
|
||||||
|
elseif ident then
|
||||||
|
local after = duffle.skip_ws_and_cmt(source, ident_end)
|
||||||
|
if source:sub(after, after) == "(" then
|
||||||
|
local _, after_p = duffle.read_parens(source, after)
|
||||||
|
after = duffle.skip_ws_and_cmt(source, after_p or after)
|
||||||
|
if source:sub(after, after) == "{" then
|
||||||
|
local body, after_b = duffle.read_braces(source, after)
|
||||||
|
local names = collect_tb_emits(body or "", addrs)
|
||||||
|
if #names > 0 then
|
||||||
|
chains[#chains + 1] = names
|
||||||
|
end
|
||||||
|
pos = after_b or (after + 1)
|
||||||
|
else
|
||||||
|
pos = after
|
||||||
|
end
|
||||||
|
else
|
||||||
|
pos = ident_end
|
||||||
|
end
|
||||||
|
else
|
||||||
|
pos = pos + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return chains
|
||||||
|
end
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- The single source walker
|
-- The single source walker
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -1930,6 +2350,7 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
|||||||
raw_atoms = {},
|
raw_atoms = {},
|
||||||
binds = {},
|
binds = {},
|
||||||
atom_infos = {},
|
atom_infos = {},
|
||||||
|
component_atom_infos = {},
|
||||||
macros = {},
|
macros = {},
|
||||||
-- Raw marker evidence for annotation validation. The `debug_skip` boolean
|
-- Raw marker evidence for annotation validation. The `debug_skip` boolean
|
||||||
-- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip.
|
-- is stamped on the declaration record itself; the projection lives on AtomEntry.debug_skip.
|
||||||
@@ -1954,6 +2375,12 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
|||||||
-- typedef chain walking (cycle-guarded, depth <= 8), and struct field sums.
|
-- typedef chain walking (cycle-guarded, depth <= 8), and struct field sums.
|
||||||
-- See `propagate_type_sizes()` below.
|
-- See `propagate_type_sizes()` below.
|
||||||
type_name_registry = {},
|
type_name_registry = {},
|
||||||
|
reg_use_schemas = {},
|
||||||
|
tape_chains = {},
|
||||||
|
_addrs = {},
|
||||||
|
_chain = nil,
|
||||||
|
_brace_depth = 0,
|
||||||
|
reg_use_errors = {},
|
||||||
-- Shared `R_*_Code -> integer code` registry
|
-- Shared `R_*_Code -> integer code` registry
|
||||||
-- (passed in from M.run pass 1; same reference so preprocessor intercept writes are visible to the enum-value resolver).
|
-- (passed in from M.run pass 1; same reference so preprocessor intercept writes are visible to the enum-value resolver).
|
||||||
-- Stripped from `src.scan` before return.
|
-- Stripped from `src.scan` before return.
|
||||||
@@ -1987,6 +2414,48 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
|||||||
local parser = DECL_PARSERS[ident]
|
local parser = DECL_PARSERS[ident]
|
||||||
if parser then
|
if parser then
|
||||||
pos = parser(source, pos, ident_end, line_of, out)
|
pos = parser(source, pos, ident_end, line_of, out)
|
||||||
|
elseif ident == "addrs" then
|
||||||
|
local after = duffle.skip_ws_and_cmt(source, ident_end)
|
||||||
|
if source:sub(after, after) == "[" then
|
||||||
|
local inner, after_br = duffle.read_brackets(source, after)
|
||||||
|
local idx = inner and tonumber(duffle.trim(inner))
|
||||||
|
after_br = duffle.skip_ws_and_cmt(source, after_br or after)
|
||||||
|
if idx and source:sub(after_br, after_br) == "=" then
|
||||||
|
local rhs = duffle.skip_ws_and_cmt(source, after_br + 1)
|
||||||
|
local rhs_ident = duffle.read_ident(source, rhs)
|
||||||
|
if rhs_ident then out._addrs[idx] = rhs_ident end
|
||||||
|
pos = rhs
|
||||||
|
else
|
||||||
|
pos = after_br or (after + 1)
|
||||||
|
end
|
||||||
|
else
|
||||||
|
pos = ident_end
|
||||||
|
end
|
||||||
|
elseif ident == "tb_emit_" or ident == "tb_emit" then
|
||||||
|
local after = duffle.skip_ws_and_cmt(source, ident_end)
|
||||||
|
if source:sub(after, after) == "(" then
|
||||||
|
local inner, after_p = duffle.read_parens(source, after)
|
||||||
|
local name
|
||||||
|
if ident == "tb_emit_" then
|
||||||
|
name = duffle.trim(inner or ""):match("^([%w_]+)")
|
||||||
|
else
|
||||||
|
local args = duffle.split_top_level_commas(inner or "")
|
||||||
|
local last = duffle.trim(args[#args] or "")
|
||||||
|
local idx = last:match("^addrs%s*%[%s*(%d+)%s*%]$")
|
||||||
|
if idx then
|
||||||
|
name = out._addrs[tonumber(idx)]
|
||||||
|
else
|
||||||
|
name = last:match("([%w_]+)$")
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if name then
|
||||||
|
out._chain = out._chain or {}
|
||||||
|
out._chain[#out._chain + 1] = name
|
||||||
|
end
|
||||||
|
pos = after_p or (after + 1)
|
||||||
|
else
|
||||||
|
pos = ident_end
|
||||||
|
end
|
||||||
else
|
else
|
||||||
-- Unsupported identifiers follow the unrelated-token path. If a
|
-- Unsupported identifiers follow the unrelated-token path. If a
|
||||||
-- pending marker is still open, consume it so it cannot drift to a
|
-- pending marker is still open, consume it so it cannot drift to a
|
||||||
@@ -2005,12 +2474,22 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
|||||||
else
|
else
|
||||||
local markers = out.debug_skip_markers
|
local markers = out.debug_skip_markers
|
||||||
local marker = markers[#markers]
|
local marker = markers[#markers]
|
||||||
if marker and marker.pending and marker.proc_prelude then
|
|
||||||
local c = source:sub(pos, pos)
|
local c = source:sub(pos, pos)
|
||||||
|
if marker and marker.pending and marker.proc_prelude then
|
||||||
if c == "{" or c == ";" then
|
if c == "{" or c == ";" then
|
||||||
attach_debug_skip_marker(out, "unrelated")
|
attach_debug_skip_marker(out, "unrelated")
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
if c == "{" then
|
||||||
|
out._brace_depth = out._brace_depth + 1
|
||||||
|
elseif c == "}" then
|
||||||
|
if out._brace_depth == 1 and out._chain and #out._chain > 0 then
|
||||||
|
out.tape_chains[#out.tape_chains + 1] = out._chain
|
||||||
|
end
|
||||||
|
out._chain = nil
|
||||||
|
out._brace_depth = out._brace_depth - 1
|
||||||
|
if out._brace_depth < 0 then out._brace_depth = 0 end
|
||||||
|
end
|
||||||
pos = pos + 1
|
pos = pos + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -2021,6 +2500,12 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
|||||||
-- Runs AFTER the source walk so all typedef / Struct_ / Enum_ declarations have been parsed into `out.type_name_registry`.
|
-- Runs AFTER the source walk so all typedef / Struct_ / Enum_ declarations have been parsed into `out.type_name_registry`.
|
||||||
-- Mutates each entry's `byte_size` field in place; fields with pointer_depth > 0 already carry byte_size = 4 from parse time and are unaffected.
|
-- Mutates each entry's `byte_size` field in place; fields with pointer_depth > 0 already carry byte_size = 4 from parse time and are unaffected.
|
||||||
propagate_type_sizes(out)
|
propagate_type_sizes(out)
|
||||||
|
if out._chain and #out._chain > 0 then
|
||||||
|
out.tape_chains[#out.tape_chains + 1] = out._chain
|
||||||
|
end
|
||||||
|
out._addrs = nil
|
||||||
|
out._chain = nil
|
||||||
|
out._brace_depth = nil
|
||||||
|
|
||||||
return out
|
return out
|
||||||
end
|
end
|
||||||
@@ -2183,16 +2668,21 @@ local function merge_corpus_registries(corpus)
|
|||||||
corpus.atom_ctxs = corpus.atom_ctxs or {}
|
corpus.atom_ctxs = corpus.atom_ctxs or {}
|
||||||
corpus.atom_phases = corpus.atom_phases or {}
|
corpus.atom_phases = corpus.atom_phases or {}
|
||||||
corpus.atom_infos = corpus.atom_infos or {}
|
corpus.atom_infos = corpus.atom_infos or {}
|
||||||
|
corpus.component_atom_infos = corpus.component_atom_infos or {}
|
||||||
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
|
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
|
||||||
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
|
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
|
||||||
corpus.collisions = corpus.collisions or {}
|
corpus.collisions = corpus.collisions or {}
|
||||||
|
corpus.reg_use_schemas = corpus.reg_use_schemas or {}
|
||||||
|
corpus.reg_use_errors = corpus.reg_use_errors or {}
|
||||||
|
corpus.tape_chains = corpus.tape_chains or {}
|
||||||
|
|
||||||
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
||||||
-- This is safe because M.run is the only writer to these tables within a single orchestrator invocation.
|
-- This is safe because M.run is the only writer to these tables within a single orchestrator invocation.
|
||||||
for _, key in ipairs({
|
for _, key in ipairs({
|
||||||
"register_alias_registry", "type_name_registry", "binds_by_name",
|
"register_alias_registry", "type_name_registry", "binds_by_name",
|
||||||
"atoms_by_name", "atom_views", "atom_ctxs", "atom_phases",
|
"atoms_by_name", "atom_views", "atom_ctxs", "atom_phases",
|
||||||
"atom_infos", "collisions",
|
"atom_infos", "component_atom_infos", "collisions", "reg_use_schemas", "reg_use_errors",
|
||||||
|
"tape_chains",
|
||||||
}) do
|
}) do
|
||||||
corpus[key] = {}
|
corpus[key] = {}
|
||||||
end
|
end
|
||||||
@@ -2285,6 +2775,65 @@ local function merge_corpus_registries(corpus)
|
|||||||
for _, info in ipairs(scan.atom_infos or {}) do
|
for _, info in ipairs(scan.atom_infos or {}) do
|
||||||
corpus.atom_infos[#corpus.atom_infos + 1] = info
|
corpus.atom_infos[#corpus.atom_infos + 1] = info
|
||||||
end
|
end
|
||||||
|
for _, info in ipairs(scan.component_atom_infos or {}) do
|
||||||
|
corpus.component_atom_infos[#corpus.component_atom_infos + 1] = info
|
||||||
|
end
|
||||||
|
|
||||||
|
for name, schema in pairs(scan.reg_use_schemas or {}) do
|
||||||
|
if corpus.reg_use_schemas[name] == nil then
|
||||||
|
corpus.reg_use_schemas[name] = schema
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for _, err in ipairs(scan.reg_use_errors or {}) do
|
||||||
|
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
|
||||||
|
end
|
||||||
|
for _, chain in ipairs(scan.tape_chains or {}) do
|
||||||
|
corpus.tape_chains[#corpus.tape_chains + 1] = chain
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local SCHEMA_BODY_ERROR = {
|
||||||
|
reguse_malformed = true,
|
||||||
|
reguse_unknown_reg_type = true,
|
||||||
|
reguse_duplicate_alias = true,
|
||||||
|
reguse_duplicate_slot = true,
|
||||||
|
reguse_const_reg_spelling = true,
|
||||||
|
reguse_mixed_const = true,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Re-parse every RegUse_* body against the merged type_name_registry.
|
||||||
|
-- Scan-time expansion still runs when Reg_T is in the same source.
|
||||||
|
-- Missing Reg_T after merge is reguse_unknown_reg_type, not a fallback table.
|
||||||
|
local function resolve_reg_use_schemas(corpus)
|
||||||
|
local kept = {}
|
||||||
|
for _, err in ipairs(corpus.reg_use_errors or {}) do
|
||||||
|
if not SCHEMA_BODY_ERROR[err.kind] then
|
||||||
|
kept[#kept + 1] = err
|
||||||
|
end
|
||||||
|
end
|
||||||
|
corpus.reg_use_errors = kept
|
||||||
|
|
||||||
|
for name, type_entry in pairs(corpus.type_name_registry or {}) do
|
||||||
|
if name:sub(1, 7) == "RegUse_" and type_entry.body then
|
||||||
|
local fresh, errs = parse_reg_use_schema_body(
|
||||||
|
type_entry.body, corpus.type_name_registry, { require_types = true })
|
||||||
|
if fresh then
|
||||||
|
fresh.name = name
|
||||||
|
local old = corpus.reg_use_schemas[name]
|
||||||
|
fresh.source_file = (old and old.source_file) or type_entry.source_file
|
||||||
|
fresh.source_line = (old and old.source_line) or type_entry.source_line
|
||||||
|
corpus.reg_use_schemas[name] = fresh
|
||||||
|
else
|
||||||
|
corpus.reg_use_schemas[name] = nil
|
||||||
|
end
|
||||||
|
for _, err in ipairs(errs or {}) do
|
||||||
|
err.schema_name = name
|
||||||
|
err.source_file = type_entry.source_file
|
||||||
|
err.source_line = type_entry.source_line
|
||||||
|
corpus.reg_use_errors[#corpus.reg_use_errors + 1] = err
|
||||||
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -2373,6 +2922,7 @@ function M.run(ctx)
|
|||||||
|
|
||||||
-- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline).
|
-- Merge per-source scans into the corpus registries (see merge_corpus_registries for first-wins + collision discipline).
|
||||||
merge_corpus_registries(corpus)
|
merge_corpus_registries(corpus)
|
||||||
|
resolve_reg_use_schemas(corpus)
|
||||||
|
|
||||||
-- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return.
|
-- code_macros and code_macro_bodies are function-local; the GC reclaims them on M.run return.
|
||||||
return { outputs = {}, errors = {}, warnings = {} }
|
return { outputs = {}, errors = {}, warnings = {} }
|
||||||
|
|||||||
+673
-218
File diff suppressed because it is too large
Load Diff
+14
-25
@@ -91,11 +91,9 @@ if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
|
|||||||
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
|
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
|
||||||
}
|
}
|
||||||
|
|
||||||
# Download anything missing. Skip the package entirely if its dir already has
|
# Download anything missing.
|
||||||
# any contents (the legacy packages.config style means the targets file
|
# Skip the package entirely if its dir already has any contents (the legacy packages.config style means the targets file location varies per package
|
||||||
# location varies per package — `luajit.native` puts it at build/native/,
|
# — `luajit.native` puts it at build/native/, `glfw` puts it elsewhere — so we can't probe a specific path; just check whether the dir is non-empty).
|
||||||
# `glfw` puts it elsewhere — so we can't probe a specific path; just check
|
|
||||||
# whether the dir is non-empty).
|
|
||||||
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||||
foreach ($pkg in $required_packages.Values) {
|
foreach ($pkg in $required_packages.Values) {
|
||||||
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
|
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
|
||||||
@@ -122,24 +120,18 @@ foreach ($pkg in $required_packages.Values) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
# ════════════════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
|
||||||
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into
|
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick.
|
||||||
# luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick. The raw string
|
# The raw string literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384, minus 5 bytes for the `-- lualoader, ` prefix).
|
||||||
# literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or
|
# If the upstream file grows past that, trim it: remove license header, trailing whitespace, blank separators, inline comments, and shrink 4-space indent to 2-space.
|
||||||
# MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384,
|
|
||||||
# minus 5 bytes for the `-- lualoader, ` prefix). If the upstream file
|
|
||||||
# grows past that, trim it: remove license header, trailing whitespace,
|
|
||||||
# blank separators, inline comments, and shrink 4-space indent to 2-space.
|
|
||||||
# Idempotent — only writes when the raw string exceeds the limit.
|
# Idempotent — only writes when the raw string exceeds the limit.
|
||||||
# ════════════════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════════
|
||||||
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
|
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
|
||||||
if (Test-Path -LiteralPath $path_isoffi) {
|
if (Test-Path -LiteralPath $path_isoffi) {
|
||||||
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
|
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
|
||||||
$startMarker = $content.IndexOf('R"EOF(--')
|
$startMarker = $content.IndexOf('R"EOF(--')
|
||||||
$endMarker = $content.IndexOf('-- )EOF"')
|
$endMarker = $content.IndexOf('-- )EOF"')
|
||||||
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) {
|
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) { $endMarker - ($startMarker + 8) } else { -1 }
|
||||||
$endMarker - ($startMarker + 8)
|
|
||||||
} else { -1 }
|
|
||||||
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
|
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
|
||||||
if ($literalLen -gt 16379) {
|
if ($literalLen -gt 16379) {
|
||||||
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
|
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
|
||||||
@@ -168,8 +160,7 @@ if (Test-Path -LiteralPath $path_isoffi) {
|
|||||||
$newLines += $line
|
$newLines += $line
|
||||||
}
|
}
|
||||||
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
|
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
|
||||||
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) `
|
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) -replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
|
||||||
-replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
|
|
||||||
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
|
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -231,13 +222,11 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
|||||||
|
|
||||||
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
||||||
|
|
||||||
# Wipe stale *.dep files across src\mips. These cache absolute paths to the
|
# Wipe stale *.dep files across src\mips.
|
||||||
# GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
|
# These cache absolute paths to the GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
|
||||||
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
|
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
|
||||||
# `make clean` in openbios only clears its own dir — subdirs like
|
# `make clean` in openbios only clears its own dir — subdirs like common/crt0/, modplayer/, and shell/ keep their stale .dep files.
|
||||||
# common/crt0/, modplayer/, and shell/ keep their stale .dep files. Easier to
|
# Easier to just delete the lot before each build than to teach every Makefile about deepclean recursion.
|
||||||
# just delete the lot before each build than to teach every Makefile about
|
|
||||||
# deepclean recursion.
|
|
||||||
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
|
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
|
||||||
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
|
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user