mirror of
https://github.com/Ed94/pikuma_ps1.git
synced 2026-08-14 11:38:14 +00:00
Compare commits
44
Commits
57fdb9e037
..
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b695056b9a | ||
|
|
3a4d6304dd | ||
|
|
a535d381ed | ||
|
|
c447bfa877 | ||
|
|
d88e0d0487 | ||
|
|
9a6eca6047 | ||
|
|
5c9c61720f | ||
|
|
b8e31123e4 | ||
|
|
ea3e30a11e | ||
|
|
37f4712237 | ||
|
|
b699b47b28 | ||
|
|
640dab7e61 | ||
|
|
4688566767 | ||
|
|
5ebaa6e083 | ||
|
|
7f0bdefbcb | ||
|
|
d5f28b83ea | ||
|
|
3ea3e8d105 | ||
|
|
6b60cef2e8 | ||
|
|
77f19321cd | ||
|
|
2e07665920 | ||
|
|
9501bbbcc2 | ||
|
|
9b6b5535f5 | ||
|
|
7807047dc0 | ||
|
|
7daeec0ee3 | ||
|
|
3f3b691ac0 | ||
|
|
a2d79d65eb | ||
|
|
bebcc6a585 | ||
|
|
ece21ed368 | ||
|
|
144c605ad8 | ||
|
|
4afd1af0fd | ||
|
|
004a7eff19 | ||
|
|
e42c75a26a | ||
|
|
69f2c0d036 | ||
|
|
b045856dd6 | ||
|
|
68b87f1c8b | ||
|
|
917b764d95 | ||
|
|
773aa44013 | ||
|
|
2b6fe53ce8 | ||
|
|
01f7ceba7c | ||
|
|
6f2eff920d | ||
|
|
f25765a7b7 | ||
|
|
2757aa4330 | ||
|
|
748b58c5c5 | ||
|
|
6441dbc23e |
@@ -20,3 +20,4 @@ toolchain/lpeg
|
|||||||
|
|
||||||
scratch
|
scratch
|
||||||
toolchain/libpsn00b
|
toolchain/libpsn00b
|
||||||
|
scripts/pcsx_debug_helper.zip
|
||||||
|
|||||||
Vendored
+35
@@ -142,6 +142,41 @@
|
|||||||
"tbreak main",
|
"tbreak main",
|
||||||
"continue"
|
"continue"
|
||||||
]
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "Debug: Hello Camera!",
|
||||||
|
"type": "gdb",
|
||||||
|
"request": "attach",
|
||||||
|
"target": "localhost:3333",
|
||||||
|
"remote": true,
|
||||||
|
"cwd": "${workspaceRoot}",
|
||||||
|
"valuesFormatting": "parseText",
|
||||||
|
"registerLimit": "1-32",
|
||||||
|
"frameFilters": false,
|
||||||
|
"showDevDebugOutput": false,
|
||||||
|
"printCalls": false,
|
||||||
|
"stopAtConnect": true,
|
||||||
|
"gdbpath": "gdb-multiarch",
|
||||||
|
"windows": {
|
||||||
|
"gdbpath": "gdb-multiarch.exe"
|
||||||
|
},
|
||||||
|
"osx": {
|
||||||
|
"gdbpath": "gdb"
|
||||||
|
},
|
||||||
|
"executable": "${workspaceRoot}/build/hello_camera.dwarf-injected.elf",
|
||||||
|
"setupCommands": [
|
||||||
|
{ "text": "set mi-async off" },
|
||||||
|
{ "text": "set remotetimeout 0" },
|
||||||
|
{ "text": "set logging file build/gen/hello_camera.gdb.log" },
|
||||||
|
{ "text": "set logging redirect on" }
|
||||||
|
],
|
||||||
|
"autorun": [
|
||||||
|
"monitor reset shellhalt",
|
||||||
|
"load build/hello_camera.dwarf-injected.elf",
|
||||||
|
"source scripts/gdb/gdb_tape_atoms.gdb",
|
||||||
|
"tbreak main",
|
||||||
|
"continue"
|
||||||
|
]
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
#endif
|
||||||
|
enum {
|
||||||
|
bios_init_pad_2 = 0x12,
|
||||||
|
bios_start_pad_2 = 0x13,
|
||||||
|
bios_flushcache = 0x44,
|
||||||
|
bios_table_addr = 0xA0,
|
||||||
|
bios_btable_addr = 0xB0,
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
bios_pad_buffer_size = 0x22,
|
||||||
|
};
|
||||||
@@ -1,5 +1,5 @@
|
|||||||
/*
|
/*
|
||||||
* atom_dsl.h
|
* dsl.atom.h
|
||||||
* ============================================================================
|
* ============================================================================
|
||||||
*
|
*
|
||||||
* ATOM DSL: Annotation layer for tape atoms (lottes_tape.h).
|
* ATOM DSL: Annotation layer for tape atoms (lottes_tape.h).
|
||||||
@@ -57,7 +57,6 @@
|
|||||||
|
|
||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
#pragma once
|
#pragma once
|
||||||
// #include <stdint.h>
|
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
@@ -71,11 +70,31 @@
|
|||||||
/* ----------------------------------------------------------------------------
|
/* ----------------------------------------------------------------------------
|
||||||
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
* atom_reg (per-enum opt-in marker for the DWARF register-alias registry)
|
||||||
*
|
*
|
||||||
* The bare `atom_reg` token adjacent to an enum entry in mips.h / lottes_tape.h flags that alias as debug-visible for scan_source's register_alias_registry.
|
* Bare `atom_reg` token adjacent to an enum entry that alias as debug-visible for scan_source's register_alias_registry.
|
||||||
* The C preprocessor strips it to a comment so no runtime symbol is created; the Lua scanner reads the bare token.
|
* Lua scanner reads the bare token.
|
||||||
* ----------------------------------------------------------------------------*/
|
* ----------------------------------------------------------------------------*/
|
||||||
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
#define atom_reg /* atom_reg: opt the preceding enum entry into the DWARF registry */
|
||||||
|
|
||||||
|
// ----------------------------------------------------------------------------
|
||||||
|
// atom_auto_reg(atom, sym) — per-atom auto-allocated GPR binding.
|
||||||
|
// enum {
|
||||||
|
// atom_auto_reg(cube_g4_face, R_Fwdx), // expands to: R_Fwdx = R_Fwdx_Code /* atom_auto_reg: cube_g4_face */,
|
||||||
|
// atom_auto_reg(cube_g4_face, R_Eye_z) atom_type(S4), // atom_type chains after
|
||||||
|
// };
|
||||||
|
// (The macro IS the entire enum entry — no separate LHS=RHS. The `atom` scope is
|
||||||
|
// preserved in a trailing C-comment on the RHS so the Lua scanner can recover
|
||||||
|
// it after preprocessing strips the macro form. R_<Sym>_Code is resolved from gen/auto_reg.h which the .c file #include's before the enum declaration.)
|
||||||
|
#define atom_auto_reg(atom, sym) sym = sym ## _Code /* atom_auto_reg: atom */
|
||||||
|
|
||||||
|
// ----------------------------------------------------------------------------
|
||||||
|
// phase_auto_reg(phase, sym) — per-phase auto-allocated GPR binding.
|
||||||
|
// enum {
|
||||||
|
// phase_auto_reg(cube_g4, R_Temp0), // expands to: R_Temp0 = R_Temp0_Code /* phase_auto_reg: cube_g4 */,
|
||||||
|
// phase_auto_reg(cube_g4, R_Temp1),
|
||||||
|
// };
|
||||||
|
// (Same macro-as-enum-entry form as atom_auto_reg above; the `phase` scope is preserved in a trailing C-comment on the RHS for the Lua scanner to recover.)
|
||||||
|
#define phase_auto_reg(phase, sym) sym = sym ## _Code /* phase_auto_reg: phase */
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
* atom_info :
|
* atom_info :
|
||||||
* MipsAtom_(cube_tri) atom_info(
|
* MipsAtom_(cube_tri) atom_info(
|
||||||
@@ -148,12 +167,12 @@
|
|||||||
* ... body ...
|
* ... body ...
|
||||||
* atom_label(bounds_chk) ← another anchor
|
* atom_label(bounds_chk) ← another anchor
|
||||||
*
|
*
|
||||||
* atom_offset(culling, bounds_chk) ← resolved by gen/.offsets.h
|
* atom_offset(culling, bounds_chk) ← resolved by gen/offsets.h
|
||||||
*
|
*
|
||||||
* The metaprogram generates gen/atom_offsets.h with one #define with the offset value per atom_offset(F, T) call.
|
* The metaprogram generates gen/offsets.h with one #define with the offset value per atom_offset(F, T) call.
|
||||||
* The preprocessor then expands the call to the right immediate value.
|
* The preprocessor then expands the call to the right immediate value.
|
||||||
*
|
*
|
||||||
* If gen/atom_offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails.
|
* If gen/offsets.h is stale (or atom_label(name) is undefined), `atom_offset_F_T` becomes an undefined macro and the C build fails.
|
||||||
* ============================================================================*/
|
* ============================================================================*/
|
||||||
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
#define atom_offset(F, T) atom_offset_ ## F ## _ ## T
|
||||||
// atom_label is a pure annotation for the metaprogram's offset calculations.
|
// atom_label is a pure annotation for the metaprogram's offset calculations.
|
||||||
+29
-20
@@ -3,7 +3,7 @@
|
|||||||
# include "assert.h"
|
# include "assert.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member))
|
#define offset_of(type, member) cast(U8,__builtin_offsetof(type,member)) // Compiler builtin version of O_
|
||||||
#define static_assert _Static_assert
|
#define static_assert _Static_assert
|
||||||
#define typeof __typeof__
|
#define typeof __typeof__
|
||||||
#define typeof_ptr(ptr) typeof((ptr)[0])
|
#define typeof_ptr(ptr) typeof((ptr)[0])
|
||||||
@@ -28,8 +28,9 @@
|
|||||||
#define internal static // internal
|
#define internal static // internal
|
||||||
|
|
||||||
#define asm __asm__
|
#define asm __asm__
|
||||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
|
||||||
|
|
||||||
|
#define A_(data) (& data)
|
||||||
|
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||||
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
#define align_(value) __attribute__((aligned (value))) // for easy alignment
|
||||||
#define C_(type,data) ((type)(data)) // for enforced precedence
|
#define C_(type,data) ((type)(data)) // for enforced precedence
|
||||||
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
#define expect_(x, y) __builtin_expect(x, y) // so compiler knows the common path
|
||||||
@@ -43,7 +44,9 @@
|
|||||||
|
|
||||||
#define R_ restrict
|
#define R_ restrict
|
||||||
#define V_ volatile
|
#define V_ volatile
|
||||||
// Fictional, used for intiution.
|
|
||||||
|
#pragma region Fictional //, used for intiution
|
||||||
|
|
||||||
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
#define EUB_ restrict // Execute Unit Bound: Data is siloed in the ALU Register File. The Load/Store Unit is bypassed. (Route to Execution Unit. Keep in registers)
|
||||||
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
#define ISO_ restrict // Isolated Provenance: Alternative to Exu_. Guarantees electrical memory isolation,
|
||||||
// unlocking the compiler’s ability to safely pack data across multiple parallel SIMD lanes (vectorization).
|
// unlocking the compiler’s ability to safely pack data across multiple parallel SIMD lanes (vectorization).
|
||||||
@@ -67,7 +70,8 @@
|
|||||||
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
|
#define latch_load_anchor(ptr) //__atomic_load_n(ptr, ooo_anchor_)
|
||||||
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
|
#define latch_store_drain(ptr, val) //__atomic_store_n(ptr, val, ooo_drain_)
|
||||||
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
|
#define pulse_xchg_weld(ptr, val) //__atomic_exchange_n(ptr, val, ooo_weld_)
|
||||||
//end of: Fictional.
|
|
||||||
|
#pragma endreigon Fictional
|
||||||
|
|
||||||
|
|
||||||
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
|
// R_ (restrict) establishes an "Eigen" or "Proprius" mapping.
|
||||||
@@ -87,12 +91,13 @@
|
|||||||
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
#define PtrSet_(type) TypeR_(type); typedef TypeV_(type)
|
||||||
#define TSet_(type) type; typedef PtrSet_(type)
|
#define TSet_(type) type; typedef PtrSet_(type)
|
||||||
|
|
||||||
#define array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
#define Array_len(a) (U4)(sizeof(a) / sizeof(typeof((a)[0])))
|
||||||
#define array_decl(type, ...) (type[]){__VA_ARGS__}
|
#define Array_decl(type, ...) (type[]){__VA_ARGS__}
|
||||||
#define Array_sym(type,len) A ## len ## _ ## type
|
#define Array_sym(type,len) A ## len ## _ ## type
|
||||||
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
#define Array_expand(type,len) type Array_sym(type, len)[len]; typedef PtrSet_(Array_sym(type, len))
|
||||||
#define Array_(type,len) Array_expand(type,len)
|
#define Array_(type,len) Array_expand(type,len)
|
||||||
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
#define Bit_(id,b) id = (1 << b), tmpl(id,pos) = b
|
||||||
|
#define Bitmask_(b) (1u << b)
|
||||||
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
#define Enum_(underlying_type, symbol) underlying_type TSet_(symbol); enum symbol
|
||||||
#define Proc_(symbol) symbol
|
#define Proc_(symbol) symbol
|
||||||
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
#define Relative_(symbol) // Does nothing but annotate that a symbol is associated with another.
|
||||||
@@ -130,21 +135,20 @@ typedef __UINT32_TYPE__ TSet_(B4);
|
|||||||
#define u4_v(value) C_(U4 V_*, value)
|
#define u4_v(value) C_(U4 V_*, value)
|
||||||
enum { false = 0, true = 1, true_overflow, };
|
enum { false = 0, true = 1, true_overflow, };
|
||||||
|
|
||||||
#define u4_lo(value) ((value) & 0xFFFFU)
|
#define u4_lo(value) (u4_(value) & 0xFFFFU)
|
||||||
#define u4_hi(value) ((value) >> 12)
|
#define u4_hi(value) (u4_(value) >> (S_(U2) * 8))
|
||||||
|
|
||||||
typedef void Proc_(VoidFn) (void);
|
typedef void Proc_(VoidFn) (void);
|
||||||
|
|
||||||
#define kilo(n) (C_(U4, n) << 10)
|
#define Kilo_(n) (C_(U4, n) << 10)
|
||||||
#define mega(n) (C_(U4, n) << 20)
|
#define Mega_(n) (C_(U4, n) << 20)
|
||||||
#define giga(n) (C_(U4, n) << 30)
|
#define Giga_(n) (C_(U4, n) << 30)
|
||||||
#define tera(n) (C_(U4, n) << 40)
|
#define Tera_(n) (C_(U4, n) << 40)
|
||||||
|
|
||||||
#define null C_(U4, 0)
|
#define null C_(U4, 0)
|
||||||
#define nullptr C_(void*, 0)
|
#define nullptr C_(void*, 0)
|
||||||
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
#define O_(type, field) C_(U4, & C_(type*,0)->field)
|
||||||
#define OA_(type, member, idx) C_(U4, & C_(type*,0)->member[idx])
|
#define OT_(field) O_(typeof_ptr(& field), field))
|
||||||
#define OT_(field) O_(typeof_ptr(& field), filed))
|
|
||||||
#define S_(data) C_(U4, sizeof(data))
|
#define S_(data) C_(U4, sizeof(data))
|
||||||
|
|
||||||
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
#define sop_1(op,a,b) C_(U1, s1_(a) op s1_(b))
|
||||||
@@ -165,6 +169,8 @@ def_signed_ops(le, <=)
|
|||||||
#undef def_signed_ops
|
#undef def_signed_ops
|
||||||
#undef def_signed_op
|
#undef def_signed_op
|
||||||
|
|
||||||
|
// Unused, we arent' doing any C-like asm since we have the asm dsl. We'll keep the non-generics if we somehow do.
|
||||||
|
#if 0
|
||||||
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
#define def_generic_sop(op, a, ...) _Generic((a), U1: op ## _s1, U2: op ## _s2, U4: op ## _s4) (a, __VA_ARGS__)
|
||||||
#define add_s(a,b) def_generic_sop(add,a,b)
|
#define add_s(a,b) def_generic_sop(add,a,b)
|
||||||
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
#define sub_s(a,b) def_generic_sop(sub,a,b)
|
||||||
@@ -174,11 +180,12 @@ def_signed_ops(le, <=)
|
|||||||
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
#define ge_s(a,b) def_generic_sop(ge, a,b)
|
||||||
#define le_s(a,b) def_generic_sop(le, a,b)
|
#define le_s(a,b) def_generic_sop(le, a,b)
|
||||||
#undef def_generic_sop
|
#undef def_generic_sop
|
||||||
|
#endif
|
||||||
|
|
||||||
#define alignas _Alignas
|
#define alignas _Alignas
|
||||||
#define alignof _Alignof
|
#define alignof _Alignof
|
||||||
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
#define byte_pad(amount, ...) B1 glue(_PAD_, __VA_ARGS__) [amount]
|
||||||
#define pcast(type, data) (C_(type*, & (data)) [0])
|
#define C_ptr(type, data) (C_(type*, & (data)) [0])
|
||||||
|
|
||||||
#define dbg_args(...) __VA_ARGS__
|
#define dbg_args(...) __VA_ARGS__
|
||||||
|
|
||||||
@@ -193,6 +200,8 @@ def_signed_ops(le, <=)
|
|||||||
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
#define defer_info(type,expr, ...) for(type info= {__VA_ARGS__}; info.once!=1;++info.once,(expr)) // Defer with tracked state
|
||||||
|
|
||||||
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
#define do_while(cond) for (U8 once=0; once!=1 || (cond); ++once)
|
||||||
|
|
||||||
|
#define Jmp_nZero_(cond,label) if (cond) goto label;
|
||||||
#pragma endregion Control Flow & Iteration
|
#pragma endregion Control Flow & Iteration
|
||||||
|
|
||||||
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
#define span_iter(type, iter, m_begin, op, m_end) ( \
|
||||||
@@ -209,16 +218,16 @@ def_signed_ops(le, <=)
|
|||||||
typedef Span_(S4);
|
typedef Span_(S4);
|
||||||
typedef Span_(U4);
|
typedef Span_(U4);
|
||||||
|
|
||||||
#if 0
|
|
||||||
#pragma region Debug
|
#pragma region Debug
|
||||||
#define debug_trap() __builtin_debugtrap()
|
#define debug_trap() __builtin_trap()
|
||||||
#if BUILD_DEBUG
|
#if BUILD_DEBUG
|
||||||
IA_ void assert(U8 cond) { if(cond){return;} else{debug_trap(); ms_exit_process(1);} }
|
#define assert(cond) if(cond == false){debug_trap();}
|
||||||
#else
|
#else
|
||||||
#define assert(cond)
|
# ifndef assert
|
||||||
|
# include <assert.h>
|
||||||
|
# endif
|
||||||
#endif
|
#endif
|
||||||
#pragma endregion Debug
|
#pragma endregion Debug
|
||||||
#endif
|
|
||||||
|
|
||||||
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
#define GCC_OPTIMIZATION_DISABLE _Pragma("GCC push_options") _Pragma("GCC optimize(\"O0\")")
|
||||||
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
#define GCC_OPTIMIZATION_ENABLE _Pragma("GCC pop_options")
|
||||||
|
|||||||
@@ -1,150 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
#pragma once
|
|
||||||
#endif
|
|
||||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
|
||||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
|
||||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
|
||||||
|
|
||||||
#ifndef WORD_COUNT
|
|
||||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
/* ---------------------------------------------------------------------------
|
|
||||||
* MACRO ATOM Components (Reusable Assembly Components)
|
|
||||||
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
|
||||||
* ---------------------------------------------------------------------------*/
|
|
||||||
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
|
||||||
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
|
|
||||||
// - mac_yield_load() + mac_yield_tail():
|
|
||||||
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
|
|
||||||
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
|
|
||||||
#define mac_yield(...) \
|
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0) \
|
|
||||||
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
|
||||||
, jump_reg( R_AtomJmp) \
|
|
||||||
, nop
|
|
||||||
WORD_COUNT(mac_yield, 4)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_yield_load(...) \
|
|
||||||
load_word(R_AtomJmp, R_TapePtr, 0)
|
|
||||||
WORD_COUNT(mac_yield_load, 1)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_yield_tail(...) \
|
|
||||||
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
|
||||||
, jump_reg( R_AtomJmp) \
|
|
||||||
, nop
|
|
||||||
WORD_COUNT(mac_yield_tail, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
|
||||||
#define mac_load_tri_indices(...) \
|
|
||||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)) \
|
|
||||||
, load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)) \
|
|
||||||
, load_half_u(R_T2, R_FaceCursor, 2 * S_(S2))
|
|
||||||
WORD_COUNT(mac_load_tri_indices, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
|
||||||
#define mac_gte_load_tri_verts(...) \
|
|
||||||
shift_lleft(R_AT, R_T0, v3s2_byteoff) \
|
|
||||||
, add_u_self(R_AT, R_VertBase) \
|
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
|
||||||
, gte_mv_to_data_r(R_V0, C2_VXY0) \
|
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
|
||||||
, shift_lleft(R_AT, R_T1, v3s2_byteoff) \
|
|
||||||
, add_u_self(R_AT, R_VertBase) \
|
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
|
||||||
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
|
||||||
, shift_lleft(R_AT, R_T2, v3s2_byteoff) \
|
|
||||||
, add_u_self(R_AT, R_VertBase) \
|
|
||||||
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
|
||||||
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
|
||||||
, gte_mv_to_data_r(R_V0, C2_VXY2) \
|
|
||||||
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
|
||||||
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
|
||||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
|
||||||
#define mac_insert_ot_tag_f3(...) \
|
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
|
||||||
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
|
||||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
|
||||||
, load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (5 - 1) << 24 = 4 << 24 */ \
|
|
||||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
|
||||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
|
||||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
|
||||||
, shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
|
||||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
|
||||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
|
||||||
WORD_COUNT(mac_insert_ot_tag_f3, 11)
|
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
|
||||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
|
||||||
#define mac_insert_ot_tag_g4(...) \
|
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
|
||||||
, add_u_self( R_T1, R_OtBase) /* T1 = & OrderingTable[OTZ] */ \
|
|
||||||
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
|
||||||
, load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) /* V0 = (9 - 1) << 24 = 8 << 24 */ \
|
|
||||||
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
|
||||||
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
|
||||||
, store_word( R_AT, R_PrimCursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
|
||||||
, shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
|
||||||
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
|
||||||
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
|
||||||
WORD_COUNT(mac_insert_ot_tag_g4, 11)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_pack_color_word(off, cmd, r, g, b) \
|
|
||||||
load_upper_i(R_AT, (cmd) << 8 | (b)) \
|
|
||||||
, or_i_self( R_AT, ((g) << 8) | (r)) \
|
|
||||||
, store_word( R_AT, R_PrimCursor, (off))
|
|
||||||
WORD_COUNT(mac_pack_color_word, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_format_f3_color(r, g, b) \
|
|
||||||
mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
|
||||||
WORD_COUNT(mac_format_f3_color, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
|
||||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
|
||||||
#define mac_gte_store_f3(...) \
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)) \
|
|
||||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)) \
|
|
||||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2))
|
|
||||||
WORD_COUNT(mac_gte_store_f3, 3)
|
|
||||||
|
|
||||||
#define mac_format_g4_color(r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
|
||||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
|
||||||
, mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1) \
|
|
||||||
, mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2) \
|
|
||||||
, mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3)
|
|
||||||
WORD_COUNT(mac_format_g4_color, 12)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
|
||||||
* G4 triangle portion to p0/p1/p2.
|
|
||||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
|
||||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
|
||||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
|
||||||
#define mac_gte_store_g4_p012(...) \
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)) \
|
|
||||||
, gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)) \
|
|
||||||
, gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2))
|
|
||||||
WORD_COUNT(mac_gte_store_g4_p012, 3)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
|
||||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
|
||||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
|
||||||
*/
|
|
||||||
#define mac_gte_store_g4_p3(...) \
|
|
||||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3))
|
|
||||||
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
|
||||||
|
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
|
||||||
// Source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#pragma region lottes_tape
|
|
||||||
|
|
||||||
|
|
||||||
#pragma endregion lottes_tape
|
|
||||||
|
|
||||||
@@ -0,0 +1,308 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\duffle/
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
|
||||||
|
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||||
|
|
||||||
|
#ifndef WORD_COUNT
|
||||||
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
#endif
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
/* ---------------------------------------------------------------------------
|
||||||
|
* MACRO ATOM Components (Reusable Assembly Components)
|
||||||
|
* These do NOT yield. They are expanded inline inside Tape Atoms.
|
||||||
|
* ---------------------------------------------------------------------------*/
|
||||||
|
// The 'Yield' sequence for Tape Atoms (mac_yield).
|
||||||
|
// - mac_yield() is the safe default for atom-endings: 4 words, BD-slot of jr is mandatory nop.
|
||||||
|
// - mac_yield_load() + mac_yield_tail():
|
||||||
|
// - unconditional branch: mac_yield_load fills the branch's BD-slot (replaces a nop);
|
||||||
|
// - mac_yield_tail runs at the branch target (does NOT re-load R_AtomJmp).
|
||||||
|
#define mac_yield(...) \
|
||||||
|
load_word(R_AtomJmp, R_TapePtr, 0) \
|
||||||
|
, add_ui_self( R_TapePtr, S_(MipsCode)) \
|
||||||
|
, jump_reg( R_AtomJmp) \
|
||||||
|
, nop
|
||||||
|
WORD_COUNT(mac_yield, 4)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_yield_load(...) \
|
||||||
|
load_word(R_AtomJmp, R_TapePtr, 0)
|
||||||
|
WORD_COUNT(mac_yield_load, 1)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_yield_tail(...) \
|
||||||
|
add_ui_self(R_TapePtr, S_(MipsCode)) \
|
||||||
|
, jump_reg( R_AtomJmp) \
|
||||||
|
, nop
|
||||||
|
WORD_COUNT(mac_yield_tail, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
||||||
|
load_half( rs_x, r_base, offset + O_(V3_S2,x)) \
|
||||||
|
, load_half( rs_y, r_base, offset + O_(V3_S2,y))
|
||||||
|
WORD_COUNT(mac_load_v2s2, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
|
||||||
|
store_half(rt_x, base, offset + O_(V2_S2,x)) \
|
||||||
|
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
||||||
|
WORD_COUNT(mac_store_v2s2, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_v3s4(rs_x, rs_y, rs_z, r_base, offset) \
|
||||||
|
load_word( rs_x, r_base, offset + O_(V3_S4,x)) \
|
||||||
|
, load_word( rs_y, r_base, offset + O_(V3_S4,y)) \
|
||||||
|
, load_word( rs_z, r_base, offset + O_(V3_S4,z))
|
||||||
|
WORD_COUNT(mac_load_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_v3s4(rt_x, rt_y, rt_z, base, offset) \
|
||||||
|
store_word(rt_x, base, offset + O_(V3_S4,x)) \
|
||||||
|
, store_word(rt_y, base, offset + O_(V3_S4,y)) \
|
||||||
|
, store_word(rt_z, base, offset + O_(V3_S4,z))
|
||||||
|
WORD_COUNT(mac_store_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_sub_v3s4(rds_x, rds_y, rds_z, rt_x, rt_y, rt_z) \
|
||||||
|
sub_s(rds_x, rds_x, rt_x) \
|
||||||
|
, sub_s(rds_y, rds_y, rt_y) \
|
||||||
|
, sub_s(rds_z, rds_z, rt_z)
|
||||||
|
WORD_COUNT(mac_sub_v3s4, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
||||||
|
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
||||||
|
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
|
||||||
|
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
|
||||||
|
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
||||||
|
WORD_COUNT(mac_store_rects2, 4)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_word_imm(dst, imm) \
|
||||||
|
load_upper_i(dst, u4_hi(imm)) \
|
||||||
|
, or_i_self( dst, u4_lo(imm))
|
||||||
|
WORD_COUNT(mac_load_word_imm, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_load_tri_indices(r_face_cusor, r_i0, r_i1, r_i2) \
|
||||||
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)) \
|
||||||
|
, load_half_u(r_i1, r_face_cusor, 1 * S_(S2)) \
|
||||||
|
, load_half_u(r_i2, r_face_cusor, 2 * S_(S2))
|
||||||
|
WORD_COUNT(mac_load_tri_indices, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_store_f3(r_primitive_cursor) \
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)) \
|
||||||
|
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)) \
|
||||||
|
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2))
|
||||||
|
WORD_COUNT(mac_gte_store_f3, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_load_tri_verts(r_vert_base, r_v0, r_v1, r_v2) \
|
||||||
|
shift_lleft(R_AT, r_v0, v3s2_byteoff) \
|
||||||
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
|
, gte_mv_to_data_r(R_V0, C2_VXY0) \
|
||||||
|
, gte_mv_to_data_r(R_V1, C2_VZ0) \
|
||||||
|
, shift_lleft(R_AT, r_v1, v3s2_byteoff) \
|
||||||
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
|
, gte_mv_to_data_r(R_V0, C2_VXY1) \
|
||||||
|
, gte_mv_to_data_r(R_V1, C2_VZ1) \
|
||||||
|
, shift_lleft(R_AT, r_v2, v3s2_byteoff) \
|
||||||
|
, add_u_self(R_AT, r_vert_base) \
|
||||||
|
, load_word(R_V0, R_AT, O_(V3_S2,x)) \
|
||||||
|
, load_word(R_V1, R_AT, O_(V3_S2,z)) \
|
||||||
|
, gte_mv_to_data_r(R_V0, C2_VXY2) \
|
||||||
|
, gte_mv_to_data_r(R_V1, C2_VZ2)
|
||||||
|
WORD_COUNT(mac_gte_load_tri_verts, 18)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_store_g4_p012(r_primitive_cursor) \
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)) \
|
||||||
|
, gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)) \
|
||||||
|
, gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2))
|
||||||
|
WORD_COUNT(mac_gte_store_g4_p012, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_store_g4_p3(r_primitive_cursor) \
|
||||||
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3))
|
||||||
|
WORD_COUNT(mac_gte_store_g4_p3, 1)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_sqr_v3(r_sx, r_sy, r_sz, r_sq_x, r_sq_y, r_sq_z) \
|
||||||
|
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop) \
|
||||||
|
, gte_mv_from_data_r(r_sq_x, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(r_sq_y, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(r_sq_z, C2_MAC3)
|
||||||
|
WORD_COUNT(mac_gte_sqr_v3, 8)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop_slot) \
|
||||||
|
gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||||
|
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||||
|
, nop_slot \
|
||||||
|
, gte_cmdw_sqr
|
||||||
|
WORD_COUNT(mac_gte_sqr_v3s4, 5)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gte_gpf_scale(r_sx, r_sy, r_sz, r_recip_est, r_shift, r_dx, r_dy, r_dz) \
|
||||||
|
gte_mv_to_data_r(r_recip_est, C2_IR0) \
|
||||||
|
, gte_mv_to_data_r(r_sx, C2_IR1) \
|
||||||
|
, gte_mv_to_data_r(r_sy, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(r_sz, C2_IR3) \
|
||||||
|
, nop2 /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */ \
|
||||||
|
, gte_cmdw_gpf \
|
||||||
|
, gte_mv_from_data_r(r_dx, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(r_dy, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(r_dz, C2_MAC3) \
|
||||||
|
, shift_aright_var(r_dx, r_dx, r_shift) \
|
||||||
|
, shift_aright_var(r_dy, r_dy, r_shift) \
|
||||||
|
, shift_aright_var(r_dz, r_dz, r_shift)
|
||||||
|
WORD_COUNT(mac_gte_gpf_scale, 13)
|
||||||
|
|
||||||
|
#define mac_trans_mt3s3s4(r_mtx, r_off, r_t0, r_t1, r_t2) \
|
||||||
|
load_word(r_t0, r_off, O_(V3_S4,x)) \
|
||||||
|
, load_word(r_t1, r_off, O_(V3_S4,y)) \
|
||||||
|
, load_word(r_t2, r_off, O_(V3_S4,z)) \
|
||||||
|
, store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])) \
|
||||||
|
, store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])) \
|
||||||
|
, store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2]))
|
||||||
|
WORD_COUNT(mac_trans_mt3s3s4, 6)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_lzcr_round_even_half_shift(r_shift, r_mag_sq, r_mag_sq_copy) \
|
||||||
|
and_i(r_shift, r_shift, gte_lzcr_even_mask) \
|
||||||
|
, or_u(r_mag_sq_copy, r_mag_sq, 0) \
|
||||||
|
, li_s(r_mag_sq, 31) \
|
||||||
|
, sub_s(r_mag_sq, r_mag_sq, r_shift) \
|
||||||
|
, shift_aright(r_mag_sq, r_mag_sq, 1)
|
||||||
|
WORD_COUNT(mac_lzcr_round_even_half_shift, 5)
|
||||||
|
|
||||||
|
#define mac_shift_aright_var_v3(rd_v0, rd_v1, rd_v2, rs_v0, rs_v1, rs_v2, r_shift) \
|
||||||
|
shift_aright_var(rd_v0, rs_v0, r_shift) \
|
||||||
|
, shift_aright_var(rd_v1, rs_v1, r_shift) \
|
||||||
|
, shift_aright_var(rd_v2, rs_v2, r_shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3, 3)
|
||||||
|
|
||||||
|
#define mac_shift_aright_var_v3_self(rds_v0, rds_v1, rds_v2, r_shift) \
|
||||||
|
shift_aright_var(rds_v0, rds_v0, r_shift) \
|
||||||
|
, shift_aright_var(rds_v1, rds_v1, r_shift) \
|
||||||
|
, shift_aright_var(rds_v2, rds_v2, r_shift)
|
||||||
|
WORD_COUNT(mac_shift_aright_var_v3_self, 3)
|
||||||
|
|
||||||
|
#define mac_gte_general_purpose_interopolation(to_ir0, to_ir1, to_ir2, to_ir3, fr_mac1, fr_mac2, fr_mac3, nop_slot1, nop_slot2) \
|
||||||
|
gte_mv_to_data_r(to_ir0, C2_IR0) \
|
||||||
|
, gte_mv_to_data_r(to_ir1, C2_IR1) /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */ \
|
||||||
|
, gte_mv_to_data_r(to_ir2, C2_IR2) \
|
||||||
|
, gte_mv_to_data_r(to_ir3, C2_IR3) /* IR3 = src.z (reloaded) */ \
|
||||||
|
, LdSlot_ nop_slot1 \
|
||||||
|
, LdSlot_ nop_slot2 \
|
||||||
|
, gte_cmdw_gpf \
|
||||||
|
, gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||||
|
WORD_COUNT(mac_gte_general_purpose_interopolation, 10)
|
||||||
|
|
||||||
|
#define mac_gte_mv_from_data_r_mac123(fr_mac1, fr_mac2, fr_mac3) \
|
||||||
|
gte_mv_from_data_r(fr_mac1, C2_MAC1) \
|
||||||
|
, gte_mv_from_data_r(fr_mac2, C2_MAC2) \
|
||||||
|
, gte_mv_from_data_r(fr_mac3, C2_MAC3)
|
||||||
|
WORD_COUNT(mac_gte_mv_from_data_r_mac123, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
||||||
|
mac_load_word_imm(reg_transfer, cmd) \
|
||||||
|
, store_word( reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_gcmd_push, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_store_rgb8(rr, rg, rb, base, offset) \
|
||||||
|
store_byte(rr, base, offset + O_(RGB8,r)) \
|
||||||
|
, store_byte(rg, base, offset + O_(RGB8,g)) \
|
||||||
|
, store_byte(rb, base, offset + O_(RGB8,b))
|
||||||
|
WORD_COUNT(mac_store_rgb8, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pack_color_word(r_base, off, cmd, r, g, b) \
|
||||||
|
load_upper_i(R_AT, (cmd) << 8 | (b)) \
|
||||||
|
, or_i_self( R_AT, ((g) << 8) | (r)) \
|
||||||
|
, store_word( R_AT, r_base, (off))
|
||||||
|
WORD_COUNT(mac_pack_color_word, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_format_f3_color(r_base, r, g, b) \
|
||||||
|
mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b)
|
||||||
|
WORD_COUNT(mac_format_f3_color, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_format_g4_color(r_prim_cursor, r0, g0, b0, r1, g1, b1, r2, g2, b2, r3, g3, b3) \
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0) \
|
||||||
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1) \
|
||||||
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2) \
|
||||||
|
, mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3)
|
||||||
|
WORD_COUNT(mac_format_g4_color, 12)
|
||||||
|
|
||||||
|
#define mac_insert_ot_tag(r_ot_base, r_prim_cursor, poly_size) \
|
||||||
|
shift_lleft( R_T1, R_T1, S_(U4)/2) /* T1 = otz * S_(U4) (otz arg is implicit R_T1) */ \
|
||||||
|
, add_u_self( R_T1, r_ot_base) /* T1 = & OrderingTable[OTZ] */ \
|
||||||
|
, load_word( R_AT, R_T1, O_(PolyTag,code)) /* AT = old_ot_head */ \
|
||||||
|
, load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits) \
|
||||||
|
, mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)) /* Strip upper 8 bits (length from prev cell) → keep only low 24 */ \
|
||||||
|
, or_u( R_AT, R_AT, R_V0) /* Merge length */ \
|
||||||
|
, store_word( R_AT, r_prim_cursor, O_(PolyTag,code)) /* prim->tag = packed(prim_length, old_addr) */ \
|
||||||
|
, shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)) /* AT = (prim_length << 24) | old_addr */ \
|
||||||
|
, shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)) \
|
||||||
|
, store_word( R_AT, R_T1, O_(PolyTag,code)) /* OrderingTable[OTZ] = PrimCursor */
|
||||||
|
WORD_COUNT(mac_insert_ot_tag, 11)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_set_centered_axes(state, scratch) \
|
||||||
|
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF) \
|
||||||
|
, or_i_self( scratch, PadAxis_Centered & 0xFFFF) /* mac_load_word_imm(scratch, PadAxis_Centered), */ \
|
||||||
|
, store_word( scratch, state, O_(PadState,axes))
|
||||||
|
WORD_COUNT(mac_pad_set_centered_axes, 3)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_set_id_byte(state, r_id, id_value) \
|
||||||
|
add_ui( r_id, R_0, id_value) \
|
||||||
|
, store_byte(r_id, state, O_(PadState,id))
|
||||||
|
WORD_COUNT(mac_pad_set_id_byte, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_set_status(r_tmp, r_state, pad_status) \
|
||||||
|
add_ui( r_tmp, R_0, pad_status) \
|
||||||
|
, store_word(r_tmp, r_state, O_(PadState,status))
|
||||||
|
WORD_COUNT(mac_pad_set_status, 2)
|
||||||
|
|
||||||
|
/* atom_dbg_skip */
|
||||||
|
#define mac_pad_store_inverted_buttons(r_buttons, r_pad_state) \
|
||||||
|
nor_u( r_buttons, r_buttons, R_0) \
|
||||||
|
, store_half( r_buttons, r_pad_state, O_(PadState,buttons))
|
||||||
|
WORD_COUNT(mac_pad_store_inverted_buttons, 2)
|
||||||
|
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\duffle\
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\word_count.metadata.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\memory.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gcc_asm.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\dsl.atom.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\lottes_tape.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\bios.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\math.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\mips.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gte.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\gp.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\pad.atom.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\duffle\psyq.atom.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region duffle
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: normalize_v3s4 (56 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_aligned_done_srav_path 3
|
||||||
|
#define _atom_offset_srav_path_aligned_done 4
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_aligned_done_srav_path = _atom_offset_aligned_done_srav_path,
|
||||||
|
atom_offset_srav_path_aligned_done = _atom_offset_srav_path_aligned_done,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: pad_bios_snapshot (84 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_snap_root_skip_disconnected 10
|
||||||
|
#define _atom_offset_disconnected_snap_end 65
|
||||||
|
#define _atom_offset_case_2_id_dispatch 9
|
||||||
|
#define _atom_offset_pending_snap_end 54
|
||||||
|
#define _atom_offset_id_dispatch_try_analog_stick 12
|
||||||
|
#define _atom_offset_id_dispatch_snap_end 40
|
||||||
|
#define _atom_offset_try_analog_stick_try_analog_pad 13
|
||||||
|
#define _atom_offset_analog_stick_snap_end 25
|
||||||
|
#define _atom_offset_try_analog_pad_try_unsupported 12
|
||||||
|
#define _atom_offset_analog_pad_snap_end 10
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||||
|
atom_offset_disconnected_snap_end = _atom_offset_disconnected_snap_end,
|
||||||
|
atom_offset_case_2_id_dispatch = _atom_offset_case_2_id_dispatch,
|
||||||
|
atom_offset_pending_snap_end = _atom_offset_pending_snap_end,
|
||||||
|
atom_offset_id_dispatch_try_analog_stick = _atom_offset_id_dispatch_try_analog_stick,
|
||||||
|
atom_offset_id_dispatch_snap_end = _atom_offset_id_dispatch_snap_end,
|
||||||
|
atom_offset_try_analog_stick_try_analog_pad = _atom_offset_try_analog_stick_try_analog_pad,
|
||||||
|
atom_offset_analog_stick_snap_end = _atom_offset_analog_stick_snap_end,
|
||||||
|
atom_offset_try_analog_pad_try_unsupported = _atom_offset_try_analog_pad_try_unsupported,
|
||||||
|
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion duffle
|
||||||
|
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "dsl.h"
|
||||||
|
# include "gp.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(gp_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gcmd_push(AtomBuilder_R ab, U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_load_word_imm(reg_transfer, cmd),
|
||||||
|
store_word( reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_rgb8(AtomBuilder_R ab, U1 rr, U1 rg, U1 rb, U4 base, U4 offset)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_byte(rr, base, offset + O_(RGB8,r)),
|
||||||
|
store_byte(rg, base, offset + O_(RGB8,g)),
|
||||||
|
store_byte(rb, base, offset + O_(RGB8,b)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pack_color_word(AtomBuilder_R ab, U4 r_base, U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
||||||
|
or_i_self( R_AT, ((g) << 8) | (r)),
|
||||||
|
store_word( R_AT, r_base, (off)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_format_f3_color(AtomBuilder_R ab, U4 r_base, U1 r, U1 g, U1 b)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, { mac_pack_color_word(r_base, O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_format_g4_color(AtomBuilder_R ab, U4 r_prim_cursor,
|
||||||
|
U1 r0, U1 g0, U1 b0,
|
||||||
|
U1 r1, U1 g1, U1 b1,
|
||||||
|
U1 r2, U1 g2, U1 b2,
|
||||||
|
U1 r3, U1 g3, U1 b3)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c1), 0, r1,g1,b1),
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c2), 0, r2,g2,b2),
|
||||||
|
mac_pack_color_word(r_prim_cursor, O_(Poly_G4,c3), 0, r3,g3,b3),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list. */
|
||||||
|
I_ Slice_MipsCode ac_insert_ot_tag(AtomBuilder_R ab, U4 r_ot_base, U4 r_prim_cursor, U4 poly_size) MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
||||||
|
add_u_self( R_T1, r_ot_base), // T1 = & OrderingTable[OTZ]
|
||||||
|
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
||||||
|
load_upper_i(R_V0, (poly_size/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits),
|
||||||
|
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
||||||
|
or_u( R_AT, R_AT, R_V0), // Merge length
|
||||||
|
store_word( R_AT, r_prim_cursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
||||||
|
shift_lleft( R_AT, r_prim_cursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
||||||
|
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
||||||
|
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
+54
-55
@@ -21,7 +21,7 @@
|
|||||||
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
* 4. Semantic encoders gp0_word_poly_f3(r,g,b)
|
||||||
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
* 3. Composite encoders enc_color_word(cmd, r, g, b)
|
||||||
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
* 2. Per-field encoders enc_gp0_color_r(r), enc_gp0_color_g(g), ...
|
||||||
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_mask = 0xFF
|
* 1. Bitfield layout consts gp0_color_red_shift = 0, gp0_color_red_width = 8
|
||||||
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
* 0. Opcode IDs gp0_cmd_poly_f3 = 0x20
|
||||||
*
|
*
|
||||||
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
* Vendor mnemonics (gte_mtc2, gte_mfc2, etc.) are NOT in this header.
|
||||||
@@ -74,7 +74,7 @@ enum {
|
|||||||
* ============================================================================
|
* ============================================================================
|
||||||
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
* 8-bit GP0 opcodes (the upper byte of a primitive's first word). These are the BYTE only.
|
||||||
* NO macro body past this point uses a raw shift or raw mask.
|
* NO macro body past this point uses a raw shift or raw mask.
|
||||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT / REG_MASK convention from mips.h.
|
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention from mips.h.
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
gp0_cmd_Nop = 0x00,
|
gp0_cmd_Nop = 0x00,
|
||||||
@@ -116,21 +116,20 @@ enum {
|
|||||||
gp0_cmd_SetDrawOffset = 0xE5,
|
gp0_cmd_SetDrawOffset = 0xE5,
|
||||||
gp0_cmd_SetMaskBit = 0xE6,
|
gp0_cmd_SetMaskBit = 0xE6,
|
||||||
|
|
||||||
/* bitfield shifts / widths / masks ----
|
/* bitfield shifts / widths ----
|
||||||
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
* Generic GP0/GP1 command byte (upper 8 bits of every word sent to either port). */
|
||||||
gp0_cmd_shift = 24,
|
gp0_cmd_shift = 24,
|
||||||
gp0_cmd_width = 8,
|
gp0_cmd_width = 8,
|
||||||
gp0_cmd_mask = 0xFF,
|
|
||||||
|
|
||||||
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
/* Color word layout (lives in Poly_F3.color, Poly_G4.c0..c3, etc.):
|
||||||
* bits 31..24 = command byte
|
* bits 31..24 = command byte
|
||||||
* bits 23..16 = BLUE
|
* bits 23..16 = BLUE
|
||||||
* bits 15..08 = GREEN
|
* bits 15..08 = GREEN
|
||||||
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
* bits 07..00 = RED (PSX GPU is BGR, NOT RGB) */
|
||||||
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8, gp0_color_cmd_mask = 0xFF,
|
gp0_color_cmd_shift = 24, gp0_color_cmd_width = 8,
|
||||||
gp0_color_blue_shift = 16, gp0_color_blue_width = 8, gp0_color_blue_mask = 0xFF,
|
gp0_color_blue_shift = 16, gp0_color_blue_width = 8,
|
||||||
gp0_color_green_shift = 8, gp0_color_green_width = 8, gp0_color_green_mask = 0xFF,
|
gp0_color_green_shift = 8, gp0_color_green_width = 8,
|
||||||
gp0_color_red_shift = 0, gp0_color_red_width = 8, gp0_color_red_mask = 0xFF,
|
gp0_color_red_shift = 0, gp0_color_red_width = 8,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ============================================================================
|
/* ============================================================================
|
||||||
@@ -143,12 +142,12 @@ enum {
|
|||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
|
|
||||||
/* ---- Layer 1.5: per-field encoders ---- */
|
/* ---- Layer 1.5: per-field encoders ---- */
|
||||||
#define enc_gp0_cmd(cmd) (((cmd) & gp0_cmd_mask) << gp0_cmd_shift)
|
#define enc_gp0_cmd(cmd) ((cmd) << gp0_cmd_shift)
|
||||||
|
|
||||||
#define enc_gp0_color_cmd(cmd) (((cmd) & gp0_color_cmd_mask) << gp0_color_cmd_shift)
|
#define enc_gp0_color_cmd(cmd) ((cmd) << gp0_color_cmd_shift)
|
||||||
#define enc_gp0_color_r(r) (((r) & gp0_color_red_mask) << gp0_color_red_shift)
|
#define enc_gp0_color_r(r) ((r) << gp0_color_red_shift)
|
||||||
#define enc_gp0_color_g(g) (((g) & gp0_color_green_mask) << gp0_color_green_shift)
|
#define enc_gp0_color_g(g) ((g) << gp0_color_green_shift)
|
||||||
#define enc_gp0_color_b(b) (((b) & gp0_color_blue_mask) << gp0_color_blue_shift)
|
#define enc_gp0_color_b(b) ((b) << gp0_color_blue_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: composite encoders ---- */
|
/* ---- Layer 2: composite encoders ---- */
|
||||||
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
#define enc_color_word(cmd, r, g, b) (enc_gp0_color_cmd(cmd) | enc_gp0_color_r(r) | enc_gp0_color_g(g) | enc_gp0_color_b(b))
|
||||||
@@ -211,38 +210,38 @@ enum {
|
|||||||
gp1_disp_Color24 = 0x1,
|
gp1_disp_Color24 = 0x1,
|
||||||
gp1_disp_VInterlace = 0x1,
|
gp1_disp_VInterlace = 0x1,
|
||||||
|
|
||||||
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/masks ---- */
|
/* ---- Layer 1: GP1 display-mode + range + draw-area shifts/widths ---- */
|
||||||
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2, gp1_disp_hres_mask = 0x3,
|
gp1_disp_hres_shift = 0, gp1_disp_hres_width = 2,
|
||||||
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1, gp1_disp_vres_mask = 0x1,
|
gp1_disp_vres_shift = 2, gp1_disp_vres_width = 1,
|
||||||
gp1_disp_color_shift = 4, gp1_disp_color_width = 1, gp1_disp_color_mask = 0x1,
|
gp1_disp_color_shift = 4, gp1_disp_color_width = 1,
|
||||||
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1, gp1_disp_interlace_mask = 0x1,
|
gp1_disp_interlace_shift = 5, gp1_disp_interlace_width = 1,
|
||||||
|
|
||||||
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
/* GP1 horizontal display range: bits 0..11 = X2, bits 12..23 = X1 */
|
||||||
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12, gp1_hrange_x1_mask = 0xFFF,
|
gp1_hrange_x1_shift = 12, gp1_hrange_x1_width = 12,
|
||||||
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12, gp1_hrange_x2_mask = 0xFFF,
|
gp1_hrange_x2_shift = 0, gp1_hrange_x2_width = 12,
|
||||||
|
|
||||||
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
/* GP1 vertical display range: bits 0..9 = Y2, bits 10..19 = Y1 */
|
||||||
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10, gp1_vrange_y1_mask = 0x3FF,
|
gp1_vrange_y1_shift = 10, gp1_vrange_y1_width = 10,
|
||||||
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10, gp1_vrange_y2_mask = 0x3FF,
|
gp1_vrange_y2_shift = 0, gp1_vrange_y2_width = 10,
|
||||||
|
|
||||||
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
/* GP1 draw area (top-left or bottom-right): bits 0..9 = X, bits 10..19 = Y
|
||||||
* (10-bit signed — caller pre-signs and masks with the named mask) */
|
* (10-bit signed — caller pre-signs) */
|
||||||
gp1_draw_x_shift = 0, gp1_draw_x_width = 10, gp1_draw_x_mask = 0x3FF,
|
gp1_draw_x_shift = 0, gp1_draw_x_width = 10,
|
||||||
gp1_draw_y_shift = 10, gp1_draw_y_width = 10, gp1_draw_y_mask = 0x3FF,
|
gp1_draw_y_shift = 10, gp1_draw_y_width = 10,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
/* ---- Layer 1.5: GP1 per-field encoders ---- */
|
||||||
#define enc_gp1_disp_hres(h) (((h) & gp1_disp_hres_mask) << gp1_disp_hres_shift)
|
#define enc_gp1_disp_hres(h) ((h) << gp1_disp_hres_shift)
|
||||||
#define enc_gp1_disp_vres(v) (((v) & gp1_disp_vres_mask) << gp1_disp_vres_shift)
|
#define enc_gp1_disp_vres(v) ((v) << gp1_disp_vres_shift)
|
||||||
#define enc_gp1_disp_color(c) (((c) & gp1_disp_color_mask) << gp1_disp_color_shift)
|
#define enc_gp1_disp_color(c) ((c) << gp1_disp_color_shift)
|
||||||
#define enc_gp1_disp_interlace(i) (((i) & gp1_disp_interlace_mask) << gp1_disp_interlace_shift)
|
#define enc_gp1_disp_interlace(i) ((i) << gp1_disp_interlace_shift)
|
||||||
|
|
||||||
#define enc_gp1_hrange_x1(x1) (((x1) & gp1_hrange_x1_mask) << gp1_hrange_x1_shift)
|
#define enc_gp1_hrange_x1(x1) ((x1) << gp1_hrange_x1_shift)
|
||||||
#define enc_gp1_hrange_x2(x2) (((x2) & gp1_hrange_x2_mask) << gp1_hrange_x2_shift)
|
#define enc_gp1_hrange_x2(x2) ((x2) << gp1_hrange_x2_shift)
|
||||||
#define enc_gp1_vrange_y1(y1) (((y1) & gp1_vrange_y1_mask) << gp1_vrange_y1_shift)
|
#define enc_gp1_vrange_y1(y1) ((y1) << gp1_vrange_y1_shift)
|
||||||
#define enc_gp1_vrange_y2(y2) (((y2) & gp1_vrange_y2_mask) << gp1_vrange_y2_shift)
|
#define enc_gp1_vrange_y2(y2) ((y2) << gp1_vrange_y2_shift)
|
||||||
#define enc_gp1_draw_x(x) (((x) & gp1_draw_x_mask) << gp1_draw_x_shift)
|
#define enc_gp1_draw_x(x) ((x) << gp1_draw_x_shift)
|
||||||
#define enc_gp1_draw_y(y) (((y) & gp1_draw_y_mask) << gp1_draw_y_shift)
|
#define enc_gp1_draw_y(y) ((y) << gp1_draw_y_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: GP1 composite encoders ---- */
|
/* ---- Layer 2: GP1 composite encoders ---- */
|
||||||
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
#define enc_gp1_disp_mode_word(h, v, c, i) (enc_gp0_cmd(gp1_cmd_DisplayMode) | enc_gp1_disp_hres(h) | enc_gp1_disp_vres(v) | enc_gp1_disp_color(c) | enc_gp1_disp_interlace(i))
|
||||||
@@ -555,14 +554,14 @@ typedef Struct_(Poly_GT4) {
|
|||||||
* bits 12..31 = reserved (zero)
|
* bits 12..31 = reserved (zero)
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
/* ---- Layer 1: TPage bitfield shifts / widths / masks ---- */
|
/* ---- Layer 1: TPage bitfield shifts / widths ---- */
|
||||||
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4, gp0_tpage_x_mask = 0xF,
|
gp0_tpage_x_shift = 0, gp0_tpage_x_width = 4,
|
||||||
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1, gp0_tpage_y_mask = 0x1,
|
gp0_tpage_y_shift = 4, gp0_tpage_y_width = 1,
|
||||||
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2, gp0_tpage_semi_trans_mask = 0x3,
|
gp0_tpage_semi_trans_shift = 5, gp0_tpage_semi_trans_width = 2,
|
||||||
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2, gp0_tpage_color_depth_mask = 0x3,
|
gp0_tpage_color_depth_shift = 7, gp0_tpage_color_depth_width = 2,
|
||||||
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1, gp0_tpage_dither_mask = 0x1,
|
gp0_tpage_dither_shift = 9, gp0_tpage_dither_width = 1,
|
||||||
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1, gp0_tpage_draw_to_disp_mask = 0x1,
|
gp0_tpage_draw_to_disp_shift = 10, gp0_tpage_draw_to_disp_width = 1,
|
||||||
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1, gp0_tpage_tex_disable_mask = 0x1,
|
gp0_tpage_tex_disable_shift = 11, gp0_tpage_tex_disable_width = 1,
|
||||||
|
|
||||||
/* TPage color-depth payload values (NOT bit positions — these go in
|
/* TPage color-depth payload values (NOT bit positions — these go in
|
||||||
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
* the 2-bit field at gp0_tpage_color_depth_shift). */
|
||||||
@@ -581,13 +580,13 @@ enum {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
/* ---- Layer 1.5: TPage per-field encoders. Mirrors enc_gte_sf/mx/v in gte.h. ---- */
|
||||||
#define enc_gp0_tpage_x(x) (((x) & gp0_tpage_x_mask) << gp0_tpage_x_shift)
|
#define enc_gp0_tpage_x(x) ((x) << gp0_tpage_x_shift)
|
||||||
#define enc_gp0_tpage_y(y) (((y) & gp0_tpage_y_mask) << gp0_tpage_y_shift)
|
#define enc_gp0_tpage_y(y) ((y) << gp0_tpage_y_shift)
|
||||||
#define enc_gp0_tpage_semi_trans(s) (((s) & gp0_tpage_semi_trans_mask) << gp0_tpage_semi_trans_shift)
|
#define enc_gp0_tpage_semi_trans(s) ((s) << gp0_tpage_semi_trans_shift)
|
||||||
#define enc_gp0_tpage_color_depth(c) (((c) & gp0_tpage_color_depth_mask) << gp0_tpage_color_depth_shift)
|
#define enc_gp0_tpage_color_depth(c) ((c) << gp0_tpage_color_depth_shift)
|
||||||
#define enc_gp0_tpage_dither(d) (((d) & gp0_tpage_dither_mask) << gp0_tpage_dither_shift)
|
#define enc_gp0_tpage_dither(d) ((d) << gp0_tpage_dither_shift)
|
||||||
#define enc_gp0_tpage_draw_to_disp(d) (((d) & gp0_tpage_draw_to_disp_mask) << gp0_tpage_draw_to_disp_shift)
|
#define enc_gp0_tpage_draw_to_disp(d) ((d) << gp0_tpage_draw_to_disp_shift)
|
||||||
#define enc_gp0_tpage_tex_disable(t) (((t) & gp0_tpage_tex_disable_mask) << gp0_tpage_tex_disable_shift)
|
#define enc_gp0_tpage_tex_disable(t) ((t) << gp0_tpage_tex_disable_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
/* ---- Layer 2: TPage composite encoder. Mirrors enc_gte_cmdw in gte.h ---- */
|
||||||
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
#define enc_gp0_tpage_word(x, y, semi_trans, color_depth, dither, draw_to_disp, tex_disable) \
|
||||||
@@ -617,17 +616,17 @@ typedef Struct_(TexturePage) { U4 raw; };
|
|||||||
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
* bits 24..31 = command byte — 0x20 (4bpp load) or 0x25 (8bpp load)
|
||||||
* ============================================================================ */
|
* ============================================================================ */
|
||||||
enum {
|
enum {
|
||||||
/* ---- Layer 1: CLUT bitfield shifts / widths / masks ---- */
|
/* ---- Layer 1: CLUT bitfield shifts / widths ---- */
|
||||||
gp0_clut_y_shift = 0, gp0_clut_y_width = 6, gp0_clut_y_mask = 0x3F,
|
gp0_clut_y_shift = 0, gp0_clut_y_width = 6,
|
||||||
gp0_clut_x_shift = 6, gp0_clut_x_width = 9, gp0_clut_x_mask = 0x1FF,
|
gp0_clut_x_shift = 6, gp0_clut_x_width = 9,
|
||||||
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
/* CLUT-load cmd-byte variants — the upper byte of the GP0 word. */
|
||||||
gp0_clut_cmd_Load4bpp = 0x20,
|
gp0_clut_cmd_Load4bpp = 0x20,
|
||||||
gp0_clut_cmd_Load8bpp = 0x25,
|
gp0_clut_cmd_Load8bpp = 0x25,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
/* ---- Layer 1.5: CLUT per-field encoders ---- */
|
||||||
#define enc_gp0_clut_x(x) (((x) & gp0_clut_x_mask) << gp0_clut_x_shift)
|
#define enc_gp0_clut_x(x) ((x) << gp0_clut_x_shift)
|
||||||
#define enc_gp0_clut_y(y) (((y) & gp0_clut_y_mask) << gp0_clut_y_shift)
|
#define enc_gp0_clut_y(y) ((y) << gp0_clut_y_shift)
|
||||||
|
|
||||||
/* ---- Layer 2: CLUT composite encoder ---- */
|
/* ---- Layer 2: CLUT composite encoder ---- */
|
||||||
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
#define enc_gp0_clut_word(cmd, x, y) (enc_gp0_cmd(cmd) | enc_gp0_clut_x(x) | enc_gp0_clut_y(y))
|
||||||
|
|||||||
@@ -0,0 +1,385 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gte.h"
|
||||||
|
# include "gp.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(gte_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
/* Words: 3; Loads 3 S2 indices from the face array */
|
||||||
|
FI_ Slice_MipsCode ac_load_tri_indices(AtomBuilder_R ab, U4 r_face_cusor, U4 r_i0, U4 r_i1, U4 r_i2)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half_u(r_i0, r_face_cusor, 0 * S_(S2)),
|
||||||
|
load_half_u(r_i1, r_face_cusor, 1 * S_(S2)),
|
||||||
|
load_half_u(r_i2, r_face_cusor, 2 * S_(S2)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
||||||
|
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_store_f3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_F3,p0)),
|
||||||
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_F3,p1)),
|
||||||
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_F3,p2)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
||||||
|
I_ Slice_MipsCode ac_gte_load_tri_verts(AtomBuilder_R ab, U4 r_vert_base, U4 r_v0, U4 r_v1, U4 r_v2) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_lleft(R_AT, r_v0, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
shift_lleft(R_AT, r_v1, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
||||||
|
shift_lleft(R_AT, r_v2, v3s2_byteoff), add_u_self(R_AT, r_vert_base), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
||||||
|
* G4 triangle portion to p0/p1/p2.
|
||||||
|
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
||||||
|
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
||||||
|
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_store_g4_p012(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_sw(C2_SXY0, r_primitive_cursor, O_(Poly_G4,p0)),
|
||||||
|
gte_sw(C2_SXY1, r_primitive_cursor, O_(Poly_G4,p1)),
|
||||||
|
gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p2)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
||||||
|
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
||||||
|
* SXY0 still holds v0.screen from the earlier RTPT.
|
||||||
|
*/
|
||||||
|
FI_ Slice_MipsCode ac_gte_store_g4_p3(AtomBuilder_R ab, U4 r_primitive_cursor) atom_dbg_skip MipsAtomComp_Proc_(ab, { gte_sw(C2_SXY2, r_primitive_cursor, O_(Poly_G4,p3)) })
|
||||||
|
|
||||||
|
/* ─── STAGE 1 of normalize: SQR + mfc2 MAC1/2/3 ───
|
||||||
|
* Emits squared magnitude per component (in MAC1/2/3) into caller-provided scratch regs.
|
||||||
|
* Stage 2 of normalize consumes these directly.
|
||||||
|
* Words: 8. Clobbers: IR1/2/3, MAC1/2/3. Uses gte_cmdw_sqr (sf=0, lm=1). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_sqr_v3(AtomBuilder_R ab, U4 r_sx, U4 r_sy, U4 r_sz, U4 r_sq_x, U4 r_sq_y, U4 r_sq_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
mac_gte_sqr_v3s4(r_sx, r_sy, r_sz, nop),
|
||||||
|
gte_mv_from_data_r(r_sq_x, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_sq_y, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_sq_z, C2_MAC3),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── SQR FIRE — mtc2 3 GPRs into IR1/IR2/IR3, then fire SQR. ───
|
||||||
|
* The SQR command always squares IR1/IR2/IR3 — those C2 registers are fixed.
|
||||||
|
* The GPRs holding the source vector are caller-determined.
|
||||||
|
* Words: 5 (3 mtc2 + 1 nop hazard + 1 cmd). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_sqr_v3s4(AtomBuilder_R ab, Reg r_sx, Reg r_sy, Reg r_sz, MipsCode nop_slot)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
|
nop_slot, gte_cmdw_sqr,
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── STAGE 4 of normalize: mtc2 IR0..3 + GPF + mfc2 MAC + srav finalize ───
|
||||||
|
* Reusable standalone — given an IR0 = 1/|v| estimate (typically from a sqrtbl lookup) and a shift count
|
||||||
|
* (typically (31 - LZCR)/2), multiplies IR0*IR[i] via GPF and shifts right to produce the normalized output.
|
||||||
|
* Used standalone for "scale vector by scalar".
|
||||||
|
* Words: 11. Clobbers: IR0..3, MAC1..3. Uses gte_cmdw_gpf (sf=0, lm=0). */
|
||||||
|
FI_ Slice_MipsCode ac_gte_gpf_scale(AtomBuilder_R ab,
|
||||||
|
U4 r_sx, U4 r_sy, U4 r_sz,
|
||||||
|
U4 r_recip_est, U4 r_shift,
|
||||||
|
U4 r_dx, U4 r_dy, U4 r_dz)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(r_recip_est, C2_IR0),
|
||||||
|
gte_mv_to_data_r(r_sx, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r_sy, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r_sz, C2_IR3),
|
||||||
|
nop2, /* retire IR0..IR3 → GPF input pre-fill (matches libgte 0x80016134..0x80016138) */
|
||||||
|
gte_cmdw_gpf,
|
||||||
|
gte_mv_from_data_r(r_dx, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_dy, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_dz, C2_MAC3),
|
||||||
|
shift_aright_var(r_dx, r_dx, r_shift),
|
||||||
|
shift_aright_var(r_dy, r_dy, r_shift),
|
||||||
|
shift_aright_var(r_dz, r_dz, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── TRANS MATRIX (libgte TransMatrix port) ───
|
||||||
|
* Atom component — auto-generates mac_trans_matrix Mac composer macro.
|
||||||
|
* m->t = v (struct copy; libgte's TransMatrix at 0x8001a540 is just 3 store_words, no GTE, no add).
|
||||||
|
* Uses 1 GPR (r_t1 = off value) per axis; per-axis load-delay-slot pattern.
|
||||||
|
* Words: 9. Clobbers: r_t1. */
|
||||||
|
FI_ Slice_MipsCode ac_trans_mt3s3s4(AtomBuilder_R ab
|
||||||
|
, U4 r_mtx, U4 r_off
|
||||||
|
, U4 r_t0, U4 r_t1, U4 r_t2
|
||||||
|
) MipsAtomComp_Proc_(ab, {
|
||||||
|
load_word(r_t0, r_off, O_(V3_S4,x)),
|
||||||
|
load_word(r_t1, r_off, O_(V3_S4,y)),
|
||||||
|
load_word(r_t2, r_off, O_(V3_S4,z)),
|
||||||
|
store_word(r_t0, r_mtx, O_(MT3_S2S4,t[0])),
|
||||||
|
store_word(r_t1, r_mtx, O_(MT3_S2S4,t[1])),
|
||||||
|
store_word(r_t2, r_mtx, O_(MT3_S2S4,t[2])),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* ─── LZCR ROUND EVEN + HALF-SHIFT ───
|
||||||
|
* Takes the raw LZCR leading-zero/ones count (from mfc2 C2_LZCR, range 1..32
|
||||||
|
* per PSX-SPX cop2r31) and the |v|² sum (in r_mag_sq from the MAC1+MAC2+MAC3
|
||||||
|
* add). Produces:
|
||||||
|
* r_shift ← LZCR rounded down to even (clear bit 0)
|
||||||
|
* r_mag_sq_copy ← |v|² sum (moved out of r_mag_sq before it's overwritten)
|
||||||
|
* r_mag_sq ← (31 - even_LZCR) / 2 = the final srav/GPF shift amount
|
||||||
|
*
|
||||||
|
* Rounding to even ensures (31 - LZCR) is always odd, so the >> 1 division
|
||||||
|
* is consistent — no 0.5 loss. The caller branches on LZCR < 24 to decide
|
||||||
|
* left-shift vs right-shift of r_mag_sq_copy, then saves the shift count.
|
||||||
|
*
|
||||||
|
* Note: C2_LZCR (cop2r31) is a fixed read-only C2 data register — the caller
|
||||||
|
* must read it via mfc2 from C2_LZCR; there is no register choice at the
|
||||||
|
* hardware level. Only the GPR that holds the result is caller-determined. */
|
||||||
|
FI_ Slice_MipsCode ac_lzcr_round_even_half_shift(AtomBuilder_R ab,
|
||||||
|
U4 r_shift,
|
||||||
|
U4 r_mag_sq,
|
||||||
|
U4 r_mag_sq_copy
|
||||||
|
)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
and_i(r_shift, r_shift, gte_lzcr_even_mask),
|
||||||
|
or_u(r_mag_sq_copy, r_mag_sq, 0),
|
||||||
|
li_s(r_mag_sq, 31),
|
||||||
|
sub_s(r_mag_sq, r_mag_sq, r_shift),
|
||||||
|
shift_aright(r_mag_sq, r_mag_sq, 1),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3(AtomBuilder_R ab
|
||||||
|
, Reg rd_v0, Reg rd_v1, Reg rd_v2
|
||||||
|
, Reg rs_v0, Reg rs_v1, Reg rs_v2
|
||||||
|
, Reg r_shift)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_aright_var(rd_v0, rs_v0, r_shift),
|
||||||
|
shift_aright_var(rd_v1, rs_v1, r_shift),
|
||||||
|
shift_aright_var(rd_v2, rs_v2, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_shift_aright_var_v3_self(AtomBuilder_R ab
|
||||||
|
, Reg rds_v0, Reg rds_v1, Reg rds_v2
|
||||||
|
, Reg r_shift)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
shift_aright_var(rds_v0, rds_v0, r_shift),
|
||||||
|
shift_aright_var(rds_v1, rds_v1, r_shift),
|
||||||
|
shift_aright_var(rds_v2, rds_v2, r_shift),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_gte_general_purpose_interopolation(AtomBuilder_R ab
|
||||||
|
, Reg to_ir0, Reg to_ir1, Reg to_ir2, Reg to_ir3
|
||||||
|
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
||||||
|
, MipsCode nop_slot1, MipsCode nop_slot2)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_to_data_r(to_ir0, C2_IR0),
|
||||||
|
gte_mv_to_data_r(to_ir1, C2_IR1), /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||||
|
gte_mv_to_data_r(to_ir2, C2_IR2),
|
||||||
|
gte_mv_to_data_r(to_ir3, C2_IR3), /* IR3 = src.z (reloaded) */
|
||||||
|
LdSlot_ nop_slot1,
|
||||||
|
LdSlot_ nop_slot2,
|
||||||
|
gte_cmdw_gpf,
|
||||||
|
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode gte_mv_from_data_r_mac123(AtomBuilder_R ab
|
||||||
|
, Reg fr_mac1, Reg fr_mac2, Reg fr_mac3
|
||||||
|
)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
gte_mv_from_data_r(fr_mac1, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(fr_mac2, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(fr_mac3, C2_MAC3),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
#pragma region Atom Procs
|
||||||
|
|
||||||
|
/* ─── Local copy of PSYQ's sqrtbl (1/sqrt lookup table for VectorNormal). ───
|
||||||
|
* Source: PSYQ 4.7 libgte sqrtbl at 0x800185B4 in hello_camera.elf.
|
||||||
|
* objdump -s --start-address=0x800185B4 --stop-address=0x800185F4 hello_camera.elf
|
||||||
|
* → 192 entries × 16-bit signed, in 1.12 fixed-point (max value 0x1000 = 1.0).
|
||||||
|
*
|
||||||
|
* Data is identical to the libgte original (byte-for-byte verified).
|
||||||
|
*
|
||||||
|
* ─── Per-entry semantics (decoded from libgte msc02 VectorNormal) ───
|
||||||
|
* Each entry is `1/sqrt(x)` in 1.12 fixed point (value / 4096).
|
||||||
|
* The 192 entries span 4 octaves of the input magnitude, with 48 entries per octave:
|
||||||
|
* Octave 0 (entries 0- 47): mantissa in [0x8000, 0x10000) output ~[1.000, 0.707]
|
||||||
|
* Octave 1 (entries 48- 95): mantissa in [0x10000, 0x20000) output ~[0.707, 0.500]
|
||||||
|
* Octave 2 (entries 96-143): mantissa in [0x20000, 0x40000) output ~[0.500, 0.354]
|
||||||
|
* Octave 3 (entries144-191): mantissa in [0x40000, 0x80000) output ~[0.354, 0.251]
|
||||||
|
* Within each octave, 8 sub-entries interpolate over the 8 fractional bits of the mantissa
|
||||||
|
* (the byte `(0x80 | (i mod 8))` for the lower-byte of the aligned value).
|
||||||
|
* Sampling the first value of each octave:
|
||||||
|
* [0] 0x1000 = 1.0000 ; 1 / sqrt(1.0000)
|
||||||
|
* [48] 0x0e4f = 0.8940 ; 1 / sqrt(1.2500)
|
||||||
|
* [96] 0x0d10 = 0.8164 ; 1 / sqrt(1.5000)
|
||||||
|
* [144] 0x0c0a = 0.7520 ; 1 / sqrt(1.7500)
|
||||||
|
* And representative sub-entries within octave 0 (mantissa in [0x8000, 0x8100)):
|
||||||
|
* [0] 0x1000 = 1.0000 ; 1 / sqrt(0x8000)
|
||||||
|
* [1] 0x0fe0 = 0.9922 ; 1 / sqrt(0x8100)
|
||||||
|
* [2] 0x0fc1 = 0.9846 ; 1 / sqrt(0x8200)
|
||||||
|
* [3] 0x0fa3 = 0.9773 ; 1 / sqrt(0x8300)
|
||||||
|
* [4] 0x0f85 = 0.9700 ; 1 / sqrt(0x8400)
|
||||||
|
* [5] 0x0f68 = 0.9629 ; 1 / sqrt(0x8500)
|
||||||
|
* [6] 0x0f4c = 0.9561 ; 1 / sqrt(0x8600)
|
||||||
|
* [7] 0x0f30 = 0.9492 ; 1 / sqrt(0x8700)
|
||||||
|
*
|
||||||
|
* The algorithm's `addi -64 / sll 1 / lh` selects the entry at `(aligned - 64) * 2` for the case where `aligned` has its top bit at bit 24.
|
||||||
|
* After the sllv/srav pair, `aligned` always lands in `[0x80, 0x100)`
|
||||||
|
* (with top bit at bit 24 → after `sub $aligned - 64`, the index sits in `[0x40, 0x80) * 2 = [0x80, 0x100)` bytes = entries [64, 128) within the sqrtbl).
|
||||||
|
* The earlier 64 entries (octave 0) are reached when the magnitude after shifting puts the top bit below bit 24 (the `sllv` branch),
|
||||||
|
* and the load upper_halves of the table bracket the input range.
|
||||||
|
* The later 64 entries (octaves 2-3) are the `srav` branch when the magnitude's top bit is well above bit 24.
|
||||||
|
*
|
||||||
|
* 192-entry table is reproduced verbatim from libgte (verified against libpsn00b/psxgte/vector.s:100-123 — 24 rows × 8 halfwords, last entry 0x0804). */
|
||||||
|
internal S2 const gte_normalize_sqr_tbl[192] align_(2) = {
|
||||||
|
0x1000, 0x0fe0, 0x0fc1, 0x0fa3, 0x0f85, 0x0f68, 0x0f4c, 0x0f30,
|
||||||
|
0x0f15, 0x0efb, 0x0ee1, 0x0ec7, 0x0eae, 0x0e96, 0x0e7e, 0x0e66,
|
||||||
|
0x0e4f, 0x0e38, 0x0e22, 0x0e0c, 0x0df7, 0x0de2, 0x0dcd, 0x0db9,
|
||||||
|
0x0da5, 0x0d91, 0x0d7e, 0x0d6b, 0x0d58, 0x0d45, 0x0d33, 0x0d21,
|
||||||
|
0x0d10, 0x0cff, 0x0cee, 0x0cdd, 0x0ccc, 0x0cbc, 0x0cac, 0x0c9c,
|
||||||
|
0x0c8d, 0x0c7d, 0x0c6e, 0x0c5f, 0x0c51, 0x0c42, 0x0c34, 0x0c26,
|
||||||
|
0x0c18, 0x0c0a, 0x0bfd, 0x0bef, 0x0be2, 0x0bd5, 0x0bc8, 0x0bbb,
|
||||||
|
0x0baf, 0x0ba2, 0x0b96, 0x0b8a, 0x0b7e, 0x0b72, 0x0b67, 0x0b5b,
|
||||||
|
0x0b50, 0x0b45, 0x0b39, 0x0b2e, 0x0b24, 0x0b19, 0x0b0e, 0x0b04,
|
||||||
|
0x0af9, 0x0aef, 0x0ae5, 0x0adb, 0x0ad1, 0x0ac7, 0x0abd, 0x0ab4,
|
||||||
|
0x0aaa, 0x0aa1, 0x0a97, 0x0a8e, 0x0a85, 0x0a7c, 0x0a73, 0x0a6a,
|
||||||
|
0x0a61, 0x0a59, 0x0a50, 0x0a47, 0x0a3f, 0x0a37, 0x0a2e, 0x0a26,
|
||||||
|
0x0a1e, 0x0a16, 0x0a0e, 0x0a06, 0x09fe, 0x09f6, 0x09ef, 0x09e7,
|
||||||
|
0x09e0, 0x09d8, 0x09d1, 0x09c9, 0x09c2, 0x09bb, 0x09b4, 0x09ad,
|
||||||
|
0x09a5, 0x099e, 0x0998, 0x0991, 0x098a, 0x0983, 0x097c, 0x0976,
|
||||||
|
0x096f, 0x0969, 0x0962, 0x095c, 0x0955, 0x094f, 0x0949, 0x0943,
|
||||||
|
0x093c, 0x0936, 0x0930, 0x092a, 0x0924, 0x091e, 0x0918, 0x0912,
|
||||||
|
0x090d, 0x0907, 0x0901, 0x08fb, 0x08f6, 0x08f0, 0x08eb, 0x08e5,
|
||||||
|
0x08e0, 0x08da, 0x08d5, 0x08cf, 0x08ca, 0x08c5, 0x08bf, 0x08ba,
|
||||||
|
0x08b5, 0x08b0, 0x08ab, 0x08a6, 0x08a1, 0x089c, 0x0897, 0x0892,
|
||||||
|
0x088d, 0x0888, 0x0883, 0x087e, 0x087a, 0x0875, 0x0870, 0x086b,
|
||||||
|
0x0867, 0x0862, 0x085e, 0x0859, 0x0855, 0x0850, 0x084c, 0x0847,
|
||||||
|
0x0843, 0x083e, 0x083a, 0x0836, 0x0831, 0x082d, 0x0829, 0x0824,
|
||||||
|
0x0820, 0x081c, 0x0818, 0x0814, 0x0810, 0x080c, 0x0808, 0x0804,
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ─── Full normalize (all 4 stages inline) ───
|
||||||
|
* Generic 4-stage GTE normalize (SQR → sum+LZCR → align+sqrtbl → GPF+srav).
|
||||||
|
*
|
||||||
|
* Parameterized by caller-provided scratch base + src/dst offsets.
|
||||||
|
* The caller passes r_src_offset and r_dst_offset as compile-time constants
|
||||||
|
* (typically derived from O_ macros in the caller's struct schema, e.g., `O_(CallerBundleScratch, fwd)`).
|
||||||
|
*
|
||||||
|
* This design lets any caller (with a scratch base + struct schema) use `normalize_v3s4_proc`
|
||||||
|
* without putting magic offsets in the C-side bundle helper — the offsets come from O_ macros at the call site.
|
||||||
|
*
|
||||||
|
* Body uses 9 GPRs (r_src_ptr..r_branch_tmp):
|
||||||
|
* r_src_ptr, r_dst_ptr : src/dst pointers (computed from r_scratch + caller offsets)
|
||||||
|
* r_tmp : src.x PRESERVED across stages 1-2 (NOT clobbered by mfc2 MAC2) → fed to IR1 in stage 4
|
||||||
|
* r_mac1_scratch : MAC1 result scratch (also holds aligned |v|² in stage 3)
|
||||||
|
* r_mac2_scratch : MAC2 result scratch → result.x after stage 4 sra
|
||||||
|
* r_recip_est : src.y PRESERVED across stages 1-2 → fed to IR2 in stage 4 → result.y
|
||||||
|
* r_norm : |v|² sum (stage 2) → half-shift (stage 3) → 1/|v| (stage 4 IR0)
|
||||||
|
* r_shift : shift count SAVED in stage 3 → consumed by stage 4 srav
|
||||||
|
* r_branch_tmp : src.z PRESERVED across stages 1-2 → fed to IR3 in stage 4 → result.z (also sqrtbl base addr)
|
||||||
|
*
|
||||||
|
* Atom_labels are srav_path / aligned_done
|
||||||
|
* (NOT namespaced — they're internal to this proc;
|
||||||
|
* the metaprogram's per-atom-name enum emission handles any collision across different atoms/files that share the same labels).
|
||||||
|
*
|
||||||
|
* Pool cost: 11 GPRs (well within the 9-10 caller-trash GPR budget when r_scratch is a wave-context carrier).
|
||||||
|
*
|
||||||
|
* Direct port of PSYQ libgte msc02.rel.text VectorNormal disassembly (0x800160a0..0x8001615c).
|
||||||
|
* Words: ~59 (matches libgte 0x800160a0..0x8001615c at +/- 0-2 words for BD-slot reshuffling).
|
||||||
|
* Sqrtbl: hardcoded to 0x800185B4 (libgte msc02.rel.data). Note: swapped to local.
|
||||||
|
* Pipeline: clobbers IR0..3, MAC1..3, LZCS, LZCR.
|
||||||
|
*/
|
||||||
|
/* MipsAtom_Proc_ wrapper: declares the static MipsCode[] body, then calls atombuilder_unroll(ab, ...) to copy the encoded instructions into the caller's MipsAtomBuilder arena. */
|
||||||
|
internal MipsAtom* normalize_v3s4_proc(AtomArena_R aa, U4 r_scratch /* GPR code: scratch base carrier (e.g., R_T4 = R_ResolveScratch) */
|
||||||
|
, U4 src_offset, U4 dst_offset /* GPR codes: PARAMETERIZED offsets (caller passes O_ macros) */
|
||||||
|
, Reg r_src_ptr, Reg r_dst_ptr, Reg r_tmp /* GPR codes: 3 scratch regs (src/dst computed + tmp) */
|
||||||
|
, Reg r_mac1_scratch, Reg r_mac2_scratch /* GPR codes: 2 more: MAC1/MAC2 scratch */
|
||||||
|
, Reg r_recip_est /* GPR code: |v|² sum + shift-input + sqrtbl[index] */
|
||||||
|
, Reg r_norm, Reg r_shift /* GPR codes: normalize working reg + final srav amount */
|
||||||
|
, Reg r_branch_tmp /* GPR code: scratch (shift count, branch target, lookup addr) */
|
||||||
|
)
|
||||||
|
MipsAtom_Proc_(aa, {
|
||||||
|
add_si(r_src_ptr, r_scratch, src_offset), /* r_src_ptr = &src */
|
||||||
|
// add_si(r_dst_ptr, r_scratch, r_dst_offset), /* r_dst_ptr = &dst */
|
||||||
|
|
||||||
|
/* Load src.x/y/z from r_src_ptr (caller-determined address) into r_tmp/r_recip_est/r_branch_tmp.
|
||||||
|
* r_tmp holds src.x throughout stages 1-2 — r_mac2_scratch is clobbered to MAC2 in stage 1.5 (line below). */
|
||||||
|
mac_load_v3s4(r_tmp, r_recip_est, r_branch_tmp, r_src_ptr, 0),
|
||||||
|
|
||||||
|
/* Stage 1: mtc2 src → IR1/2/3, SQR fires. */
|
||||||
|
LdSlot_ mac_gte_sqr_v3s4(r_tmp, r_recip_est, r_branch_tmp, LdSlot_ nop),
|
||||||
|
|
||||||
|
/* Stage 2: mfc2 MAC1/2/3, sum, mtc2 LZCS. */
|
||||||
|
mac_gte_mv_from_data_r_mac123(r_mac1_scratch, r_mac2_scratch, r_norm), LdSlot_ nop,
|
||||||
|
add_u_self( r_norm, r_mac1_scratch),
|
||||||
|
add_u_self( r_norm, r_mac2_scratch),
|
||||||
|
gte_mv_to_data_r( r_norm, C2_LZCS), LdSlot_ nop2,
|
||||||
|
gte_mv_from_data_r(r_shift, C2_LZCR), LdSlot_ nop,
|
||||||
|
|
||||||
|
/* Stage 3: round LZCR to even, compute half-shift, align |v|² to bit 24.
|
||||||
|
* r_norm holds |v|² sum; r_shift holds the LZCR count from mfc2.
|
||||||
|
* After the component: r_shift = even(LZCR), r_norm = half-shift, r_mac1_scratch = |v|². */
|
||||||
|
mac_lzcr_round_even_half_shift(r_shift, r_norm, r_mac1_scratch),
|
||||||
|
/* r_branch_tmp = LZCR - 24 (overwrites r_branch_tmp; src.z no longer needed after SQR) */
|
||||||
|
add_si( r_branch_tmp, r_shift, -24),
|
||||||
|
branch_lt_zero(r_branch_tmp, atom_offset(aligned_done, srav_path)), BdSlot_ nop, /* bltz → srav_path (LZCR < 24 path) */
|
||||||
|
jump_rel(atom_offset(srav_path, aligned_done)), /* b → aligned_done (LZCR >= 24 path) */
|
||||||
|
BdSlot_ shift_lleft_var(r_mac1_scratch, r_mac1_scratch, r_branch_tmp), /* src=sum (r_mac1_scratch), dst=same */
|
||||||
|
atom_label(srav_path)
|
||||||
|
li_s( r_branch_tmp, 24),
|
||||||
|
sub_s(r_branch_tmp, r_branch_tmp, r_shift),
|
||||||
|
shift_aright_var(r_mac1_scratch, r_mac1_scratch, r_branch_tmp), /* src=sum (r_mac1_scratch), dst=same */
|
||||||
|
atom_label(aligned_done)
|
||||||
|
// Save the shift count to r_shift before the next 5 instructions overwrite r_norm (the sqrtbl lookup loads 1/|v| into r_norm, which becomes IR0 in stage 4).
|
||||||
|
or_u(r_shift, r_norm, 0), /* r_shift ← shift count (preserved through stage 4) */
|
||||||
|
/* r_mac1_scratch holds |v|² aligned (top bit at bit 7). */
|
||||||
|
add_si( r_mac1_scratch, r_mac1_scratch, -64),
|
||||||
|
shift_lleft(r_mac1_scratch, r_mac1_scratch, 1),
|
||||||
|
mac_load_word_imm(r_branch_tmp, & gte_normalize_sqr_tbl), add_u_self(r_branch_tmp, r_mac1_scratch),
|
||||||
|
load_half(r_norm, r_branch_tmp, 0), /* r_norm = sqrtbl[aligned-64] = 1/|v| (IR0 in stage 4) */
|
||||||
|
|
||||||
|
/* r_branch_tmp held the sqrtbl base+index, NOT src.z. Reload src.z from scratch now that r_branch_tmp is free. */
|
||||||
|
LdSlot_ load_word(r_branch_tmp, r_src_ptr, O_(V3_S4,z)), /* r_branch_tmp = src.z (for IR3 in stage 4) */
|
||||||
|
|
||||||
|
/* Stage 4: GPF + srav finalize (r_shift = shift count, r_norm = 1/|v|). */
|
||||||
|
LdSlot_ mac_gte_general_purpose_interopolation(
|
||||||
|
r_norm,
|
||||||
|
r_tmp, /* IR1 = src.x (preserved in r_tmp — r_mac2_scratch was clobbered to MAC2 in stage 1.5) */
|
||||||
|
r_recip_est,
|
||||||
|
r_branch_tmp, /* IR3 = src.z (reloaded) */
|
||||||
|
r_mac2_scratch, r_recip_est, r_branch_tmp,
|
||||||
|
LdSlot_ add_si(r_dst_ptr, r_scratch, dst_offset), // pre-laoding destination to register here.
|
||||||
|
LdSlot_ nop
|
||||||
|
),
|
||||||
|
/* sra by r_shift = (31-LZCR)/2 (saved before sqrtbl lookup) */
|
||||||
|
mac_shift_aright_var_v3_self(r_mac2_scratch, r_recip_est, r_branch_tmp, r_shift),
|
||||||
|
/* Store result.x/y/z to r_dst_ptr (caller-determined dst address). */
|
||||||
|
mac_store_v3s4(r_mac2_scratch, r_recip_est, r_branch_tmp, r_dst_ptr, 0),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion Atom Procs
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
typedef Struct_(Binds_SetGteMT3S2S4) {
|
||||||
|
MT3_S2S4* transform;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(set_gte_mt3s2s4) atom_info(
|
||||||
|
atom_bind(Binds_SetGteMT3S2S4)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop matrix address from tape into R_T3 ($11) */
|
||||||
|
load_word(R_T3, R_TapePtr, O_(Binds_SetGteMT3S2S4,transform)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_SetGteMT3S2S4)),
|
||||||
|
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
||||||
|
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
||||||
|
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
||||||
|
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
||||||
|
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
+135
-24
@@ -161,6 +161,8 @@ enum {
|
|||||||
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
gte_cmd_nclip = 0x06, /* Normal Clipping (Backface culling) */
|
||||||
gte_cmd_op = 0x0C, /* Outer Product */
|
gte_cmd_op = 0x0C, /* Outer Product */
|
||||||
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
gte_cmd_mvmva = 0x12, /* Matrix Vector Multiply & Add (Custom math) */
|
||||||
|
gte_cmd_sqr = 0x28, /* Square vector — MAC[i] = IR[i]²; IR[i] ← MAC[i] saturated */
|
||||||
|
gte_cmd_gpf = 0x3D, /* General-purpose Interpolation — MAC[i] = IR0 * IR[i] */
|
||||||
|
|
||||||
/* --- GTE Command Bit-Field Layout ---
|
/* --- GTE Command Bit-Field Layout ---
|
||||||
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
* A GTE command word (sent to COP2 with RS=1) is laid out as:
|
||||||
@@ -171,19 +173,47 @@ enum {
|
|||||||
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
* +------------+--+-----+------+------+------+------+---+--------+----------+
|
||||||
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
* \_____ GTE_PAYLOAD _____/ \__ GTE_CMD __/
|
||||||
*
|
*
|
||||||
* Shifts/masks below are the *bit positions* and *bit widths* of each
|
* Shifts/masks below are the *bit positions* and *bit widths* of each configurable field, used by the ENC_GTE_CMD encoder.
|
||||||
* configurable field, used by the ENC_GTE_CMD encoder.
|
|
||||||
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
* Mirrors the OPCODE_SHIFT / RS_SHIFT convention used in mips.h.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
gte_shift_sf = 19, gte_width_sf = 1, gte_mask_sf = 0x1,
|
gte_shift_sf = 19, gte_width_sf = 1,
|
||||||
gte_shift_mx = 17, gte_width_mx = 2, gte_mask_mx = 0x3,
|
gte_shift_mx = 17, gte_width_mx = 2,
|
||||||
gte_shift_v = 15, gte_width_v = 2, gte_mask_v = 0x3,
|
gte_shift_v = 15, gte_width_v = 2,
|
||||||
gte_shift_cv = 13, gte_width_cv = 2, gte_mask_cv = 0x3,
|
gte_shift_cv = 13, gte_width_cv = 2,
|
||||||
gte_shift_lm = 10, gte_width_lm = 1, gte_mask_lm = 0x1,
|
gte_shift_lm = 10, gte_width_lm = 1,
|
||||||
gte_shift_cmd = 0, gte_width_cmd = 6, gte_mask_cmd = 0x3F,
|
gte_shift_cmd = 0, gte_width_cmd = 6,
|
||||||
|
|
||||||
|
/* Fake command number (bits 24-20) — IGNORED by the GTE hardware per PSX-SPX `geometrytransformationenginegte.md` line 48.
|
||||||
|
* libgte's compiler emits non-zero values in this field as a disassembly signature. */
|
||||||
|
gte_shift_fake_cmd = 20,
|
||||||
|
gte_width_fake_cmd = 5,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/* --- GTE Control Register Aliases (Pitfall 1) ---
|
||||||
|
* Three pairs of aliases map to the SAME C2 control-register slot on real silicon:
|
||||||
|
* C2[24] = gte_cr_RBK (background R) | gte_cr_OFX (screen offset X)
|
||||||
|
* C2[25] = gte_cr_GBK (background G) | gte_cr_OFY (screen offset Y)
|
||||||
|
* C2[26] = gte_cr_BBK (background B) | gte_cr_H (projection plane distance H)
|
||||||
|
* Cross-alias writes inside one atom body, or across the wave-context boundary,
|
||||||
|
* silently clobber each other. The metaprogram's check_gte_cr_alias_writes
|
||||||
|
* (CHECK_RULES row) warns about each pair per source. See
|
||||||
|
* docs/gte_reference.md §"Control-register alias table" for the silicon
|
||||||
|
* rationale and the libgte outer-product convention.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/* --- RT-matrix packed-slot convention (Pitfall 4) ---
|
||||||
|
* The silicon packs two 16-bit RT elements per 32-bit C2 slot:
|
||||||
|
* C2[2] = (RT22 << 16) | RT13 (gte_cr_RT13 writes the low half, gte_cr_RT22 writes the high half)
|
||||||
|
* C2[4] = (RT33 << 16) | RT22 (gte_cr_RT22 writes the low half — clobbers prior RT22 value if RT13 was also written)
|
||||||
|
* OP and MVMVA read D1/D2/D3 from these packed slots. The libgte outer-product
|
||||||
|
* convention (see ac_apply_matrix_lv at gte.atom.c:108-122) writes C2[2] then
|
||||||
|
* C2[4] in sequence; the SECOND write's low half is RT22, not RT13. An agent
|
||||||
|
* who writes gte_cr_RT13 then gte_cr_RT22 to the SAME source GPR clobbers the
|
||||||
|
* RT13 value. See docs/gte_reference.md §"RT-matrix packed-slot convention"
|
||||||
|
* for the canonical write pattern.
|
||||||
|
*/
|
||||||
|
|
||||||
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
/* --- GTE Control Register Indices (for ctc2/cfc2) ---
|
||||||
* Preprocessor-visible integer ids for the COP2 control register file.
|
* Preprocessor-visible integer ids for the COP2 control register file.
|
||||||
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
* Each enum value is bound to a parallel `_Code` `#define` so the preprocessor can stringify the integer (for `reg_str`/`rgcc` paths).
|
||||||
@@ -243,10 +273,10 @@ enum { _C2_OPS_ = 0
|
|||||||
* bit 1 (0x02): register class — 0 = data, 1 = control
|
* bit 1 (0x02): register class — 0 = data, 1 = control
|
||||||
* bit 2 (0x04): direction — 0 = read, 1 = write
|
* bit 2 (0x04): direction — 0 = read, 1 = write
|
||||||
*
|
*
|
||||||
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as the general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
* The values 0x00 (sub_mfc2) and 0x04 (sub_mtc2) are the same 5-bit numbers as general MIPS `cop_mf` / `cop_mt` defined in mips.h
|
||||||
* (which target the data register file on any coprocessor).
|
* (which target the data register file on any coprocessor).
|
||||||
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
* They are re-aliased here so the four-way table reads like the spec mnemonics (MFC2 / CFC2 / MTC2 / CTC2)
|
||||||
* and so the encoding lives next to its only consumer (this header).
|
* and so the encoding is next to its only consumer (this header).
|
||||||
*
|
*
|
||||||
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
* Vendor mnemonic aliases (gte_mfc2 / gte_mtc2 / gte_cfc2 / gte_ctc2) live in gte_vendor_sym.h. */
|
||||||
enum { _C2_TX_SUBS_ = 0
|
enum { _C2_TX_SUBS_ = 0
|
||||||
@@ -309,23 +339,24 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
|
|
||||||
/* GTE Command Format
|
/* GTE Command Format
|
||||||
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
* Opcode is always MIPS_OP_COP2, RS is always 1 (CO).
|
||||||
* The lower 25 bits are the GTE-specific command payload.
|
* Lower 25 bits are GTE-specific command payload.
|
||||||
*
|
*
|
||||||
* The granular `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
* The `enc_gte_<field>(x)` macros below mirror the `enc_op`/`enc_rs` pattern in mips.h:
|
||||||
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
* Each one self-masks and shifts its own field, so a caller can build up a GTE command piece by piece
|
||||||
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
* (handy for state-driven MVMVA emitters that vary one field at a time).
|
||||||
*
|
*
|
||||||
* `ENC_GTE_CMD` is the all-in-one convenience for emitting a full command word in one go.
|
* `ENC_GTE_CMD` is an all-in-one convenience for emitting a full command word.
|
||||||
* It just ORs the per-field encoders together. */
|
* It just ORs the per-field encoders together. */
|
||||||
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
#define gte_cmd_base (enc_op(op_cop2) | (1 << 25))
|
||||||
|
|
||||||
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
/* Per-field encoders. Each one does (value & mask) << shift on its own. */
|
||||||
#define enc_gte_sf(sf) (((sf) & gte_mask_sf ) << gte_shift_sf )
|
#define enc_gte_sf(sf) ((sf) << gte_shift_sf )
|
||||||
#define enc_gte_mx(mx) (((mx) & gte_mask_mx ) << gte_shift_mx )
|
#define enc_gte_mx(mx) ((mx) << gte_shift_mx )
|
||||||
#define enc_gte_v(v) (((v) & gte_mask_v ) << gte_shift_v )
|
#define enc_gte_v(v) ((v) << gte_shift_v )
|
||||||
#define enc_gte_cv(cv) (((cv) & gte_mask_cv ) << gte_shift_cv )
|
#define enc_gte_cv(cv) ((cv) << gte_shift_cv )
|
||||||
#define enc_gte_lm(lm) (((lm) & gte_mask_lm ) << gte_shift_lm )
|
#define enc_gte_lm(lm) ((lm) << gte_shift_lm )
|
||||||
#define enc_gte_cmd(cmd) (((cmd) & gte_mask_cmd) << gte_shift_cmd)
|
#define enc_gte_cmd(cmd) ((cmd) << gte_shift_cmd )
|
||||||
|
#define enc_gte_fake_cmd(x) ((x) << gte_shift_fake_cmd)
|
||||||
|
|
||||||
/* Composite: all six GTE fields + the COP2/CO base. */
|
/* Composite: all six GTE fields + the COP2/CO base. */
|
||||||
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
#define enc_gte_cmdw(sf, mx, v, cv, lm, cmd) ( \
|
||||||
@@ -363,11 +394,11 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
* (the perspective divide happens regardless of `sf`).
|
* (the perspective divide happens regardless of `sf`).
|
||||||
*
|
*
|
||||||
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
* If we emit a strictly-spec-compliant word (`sf=0`, reserved bits clear),
|
||||||
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops —
|
* PCSX-Redux's GTE checks those bits more strictly than the silicon does and RTPT silently no-ops.
|
||||||
* the floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
* The floor's screen coordinates come out as raw projection-of-rotation (Z never divided),
|
||||||
* `nclip` ends up wrong, and the triangle is culled.
|
* `nclip` ends up wrong, and the triangle is culled.
|
||||||
*
|
*
|
||||||
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern everyone has shipped for 25 years.
|
* So for RTPS and RTPT we OR-in the `0x28` "PsyQ compat" pattern to match the working bit pattern.
|
||||||
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
* NCLIP / OP / MVMVA stay spec-clean — their reserved bits really are zero in the original PsyQ source.
|
||||||
* --------------------------------------------------------------------------
|
* --------------------------------------------------------------------------
|
||||||
*/
|
*/
|
||||||
@@ -378,11 +409,92 @@ enum { _C2_TX_SUBS_ = 0
|
|||||||
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
#define gte_cmdw_nclip (gte_cmd_base | enc_gte_cmd(gte_cmd_nclip))
|
||||||
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
#define gte_cmdw_op (gte_cmd_base | enc_gte_cmd(gte_cmd_op ))
|
||||||
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
|
#define gte_cmdw_outer_product gte_cmdw_op /* "outer product" -- NOCASH/Sdk terminology */
|
||||||
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology */
|
#define gte_cmdw_wedge gte_cmdw_op /* "wedge product" -- geometric-algebra terminology.
|
||||||
|
* RGA(Lengyel): the GTE OP is a 3D signed-16-bit D x IR cross, not a generic RGA exterior product.
|
||||||
|
* The wedge alias is the 3D complement interpretation of the same 3 scalars (MAC1..MAC3). */
|
||||||
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
#define gte_cmdw_mvmva (gte_cmd_base | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA with sf=0 (no shift, full-integer), cv=3 (no translation), v=3 (IR vector input).
|
||||||
|
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = RT row · IR (full product, no >>12).
|
||||||
|
* Per PSX-SPX: SAR (sf*12) with sf=0 = SAR 0 = no shift. */
|
||||||
|
#define gte_cmdw_mvmva_sf0_ir (gte_cmd_base | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA with sf=1 (>>12 shift, 4.12 fixed-point), cv=3 (no translation), v=3 (IR): for ApplyMatrixLV.
|
||||||
|
* Reads input from IR1/2/3 (loaded via mtc2 rt, C2_IRx). MAC1/2/3 = (RT row · IR) >> 12.
|
||||||
|
* Per PSX-SPX: SAR (sf*12) with sf=1 = SAR 12 = arithmetic right-shift by 12.
|
||||||
|
* This matches the libgte C-side ApplyMatrixLV output (R*pos >> 12). */
|
||||||
|
#define gte_cmdw_mvmva_ir (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=0, mx=3 (Light matrix), v=3 (IR), cv=3 (no TR).
|
||||||
|
* For pass1 of the C11 two-pass decomposition. Reads L matrix.
|
||||||
|
* Since L matrix is typically zero, pass1 contributes 0 to the combine. */
|
||||||
|
#define gte_cmdw_mvmva_sf0_mx3_v3_cv3 (gte_cmd_base | enc_gte_sf(0) | enc_gte_cv(3) | enc_gte_v(3) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=1 (>>12), mx=3 (Light matrix), v=2 (V0), cv=0 (with TR).
|
||||||
|
* Matches the C11 ApplyMatrixLV pass 2 command word (0x49E012) exactly.
|
||||||
|
* The combine is (pass1 << 3) + pass2. */
|
||||||
|
#define gte_cmdw_mvmva_pass2_c11 (gte_cmd_base | enc_gte_sf(1) | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=0, mx=3, v=2, cv=0. Matches the C11 pass 1 command. */
|
||||||
|
#define gte_cmdw_mvmva_pass1_c11 (gte_cmd_base | enc_gte_v(2) | enc_gte_mx(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
#define gte_cmdw_mvmva_no_tr gte_cmdw_mvmva_ir
|
||||||
|
|
||||||
|
/* MVMVA pass 2 — C11 ApplyMatrixLV command.
|
||||||
|
* Decoded: op_cop2 | CO | fake_cmd=4 | sf=1 (>>12) | mx=0 (RT matrix) | v=3 (IR) | cv=3 (no translation) | lm=0 | cmd=MVMVA.
|
||||||
|
* Reads (RT row · IR) >> 12 into MAC1/2/3. Per-field composition (no opaque literal)
|
||||||
|
* keeps the bit layout visible at the call site + matches the libgte C-side byte-exact. */
|
||||||
|
#define gte_cmdw_mvmva_c11_pass2 (gte_cmd_base | enc_gte_fake_cmd(4) | enc_gte_sf(1) | enc_gte_v(3) | enc_gte_mx(0) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* MVMVA: sf=1 (>>12), mx=0 (RT matrix), v=0 (V0), cv=3 (no TR). */
|
||||||
|
#define gte_cmdw_mvmva_sf1_mx0_v0_cv3 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_v(0) | enc_gte_mx(0) | enc_gte_cmd(gte_cmd_mvmva))
|
||||||
|
|
||||||
|
/* RTPS with sf=1 (12-bit shift, no translation): matches the output of libgte's
|
||||||
|
* ApplyMatrixLV when the GTE pipeline expects R*pos >> 12. The shift produces
|
||||||
|
* values like (-270, 710, 1713) which match the C11 reference path. */
|
||||||
|
#define gte_cmdw_rtps_sf1 (gte_cmd_base | enc_gte_sf(1) | enc_gte_cv(3) | enc_gte_cmd(gte_cmd_rtps))
|
||||||
|
|
||||||
|
/* SQR / GPF cosmetic-bits compat helpers.
|
||||||
|
* Each command's `_compat` macro ORs in the `fake_cmd` field value libgte happens to emit.
|
||||||
|
* The hardware ignores these bits (per PSX-SPX line 48). */
|
||||||
|
#define gte_cmdw_sqr_fake_sig enc_gte_fake_cmd(0x0A)
|
||||||
|
#define gte_cmdw_gpf_fake_sig enc_gte_fake_cmd(0x19)
|
||||||
|
|
||||||
|
/* SQR — Square Vector.
|
||||||
|
* PSX-SPX `geometrytransformationenginegte.md` §"SQR":
|
||||||
|
* [MAC1,MAC2,MAC3] = [IR1*IR1, IR2*IR2, IR3*IR3] SHR (sf*12)
|
||||||
|
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3] (saturated to 0x7FFF when lm=1)
|
||||||
|
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x800160b0:
|
||||||
|
* 0x4AA00428 = gte_cmd_base | gte_cmdw_sqr_compat | enc_gte_lm(1) | enc_gte_cmd(0x28)
|
||||||
|
* bit 19 sf=0
|
||||||
|
* bit 10 lm=1
|
||||||
|
* bits 5-0 cmd=0x28=SQR
|
||||||
|
* bits 24-20 = 0x0A (libgte "nonsense SDK command number" signature) */
|
||||||
|
#define gte_cmdw_sqr (gte_cmd_base | enc_gte_cmd(gte_cmd_sqr) | enc_gte_lm(1) | gte_cmdw_sqr_fake_sig)
|
||||||
|
|
||||||
|
/* GPF — General-purpose Interpolation.
|
||||||
|
* PSX-SPX `geometrytransformationenginegte.md` §"GPF":
|
||||||
|
* [MAC1,MAC2,MAC3] = (([IR1,IR2,IR3] * IR0) + [MAC1,MAC2,MAC3]) SAR (sf * 12)
|
||||||
|
* [IR1,IR2,IR3] = [MAC1,MAC2,MAC3]
|
||||||
|
* Sourced verbatim from libgte msc02 VectorNormal disassembly at 0x8001613c:
|
||||||
|
* 0x4B90003D = gte_cmd_base | gte_cmdw_gpf_compat | enc_gte_cmd(0x3D)
|
||||||
|
* bit 19 sf = 0
|
||||||
|
* bit 10 lm = 0
|
||||||
|
* bits 5-0 cmd = 0x3D = GPF
|
||||||
|
* bits 24-20 = 0x19 (libgte "nonsense SDK command number" signature) */
|
||||||
|
#define gte_cmdw_gpf (gte_cmd_base | enc_gte_cmd(gte_cmd_gpf) | gte_cmdw_gpf_fake_sig)
|
||||||
|
|
||||||
|
/* Mask to round LZCR (leading-zero/ones count, range 1..32 per PSX-SPX cop2r31)
|
||||||
|
* down to even. The normalize_v3s4 half-shift logic computes (31 - LZCR) >> 1;
|
||||||
|
* clearing bit 0 ensures the subtraction result is always odd,
|
||||||
|
* so the >> 1 division is consistent (no 0.5 loss). */
|
||||||
|
enum {
|
||||||
|
gte_lzcr_even_mask = 0xFFFE, /* all bits except bit 0 */
|
||||||
|
};
|
||||||
|
|
||||||
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
#define gte_cmdw_rotate_translate_perspective_single gte_cmdw_rtps
|
||||||
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
#define gte_cmdw_rotate_translate_perspective_triple gte_cmdw_rtpt
|
||||||
|
/* RGA(Lengyel): RTPS/RTPT consume the matrix expansion of a rigid transformation (rotation matrix + translation vector) loaded into the RT/TR control registers.
|
||||||
|
* For unitized points the same result equals the motor antiproduct; the GTE executes the LA form, not a symbolic antiproduct. */
|
||||||
|
|
||||||
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
/* PsyQ compatibility bits for AVSZ3 (Bits 20, 22, 24 must be set) */
|
||||||
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
#define gte_cmdw_psyq_avsz3_compat (0x15 << 20)
|
||||||
@@ -433,7 +545,6 @@ enum {
|
|||||||
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
#define gte_lw_v2_z(base) enc_gte_lw(gte_in_v2_z, (base), GTE_Z_Offset)
|
||||||
|
|
||||||
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
/* gte_load_vN(r_ptr, base) — placeholder-punned lwc2 loaders
|
||||||
*
|
|
||||||
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
* Emits `.word` constants encoding `lwc2 $N, off(<base>)` for the chosen GTE vector register, where `<base>` is the GPR number you pass in
|
||||||
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
* (typically one of R_T4..R_T9 for the standard "3-pointer" pattern).
|
||||||
*
|
*
|
||||||
|
|||||||
+271
-217
@@ -1,36 +1,70 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# pragma once
|
# pragma once
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
|
||||||
# include "dsl.h"
|
# include "dsl.h"
|
||||||
# include "gcc_asm.h"
|
# include "gcc_asm.h"
|
||||||
# include "mips.h"
|
# include "mips.h"
|
||||||
# include "gte.h"
|
# include "gte.h"
|
||||||
# include "memory.h"
|
# include "memory.h"
|
||||||
# include "atom_dsl.h"
|
# include "dsl.atom.h"
|
||||||
# include "gen/duffle.macs.h"
|
|
||||||
# include "gen/duffle.offsets.h"
|
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
#pragma region Tape Drive
|
||||||
typedef Slice_(MipsCode);
|
/* -----------------------------------------------------------------------------------------------------------
|
||||||
|
* TAPE DRIVE ABI
|
||||||
typedef U4 const MipsAtom; // Underlying type to an array of mips asm words that must terminate with an ac_yield.
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
* Note(Ed): One of the main purposes of this codebase is to help me learn this,
|
||||||
|
* as such the information below may not* be entirely realized or finalized conceptually.
|
||||||
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
// MipsAtomComp_(ac_X) { body }
|
* This ABI and its associated legos were directly inspired by researching the work of
|
||||||
// expands to:
|
* Timothy Lottes and Onat Türkçüoğlu; along with many others. It's the simplest bootstrap of a
|
||||||
// MipsCode ac_X[] align_(4) = { body };
|
* directly executed chain of assemby arrays (Atoms) that terminate with a yield sequence to the next atom.
|
||||||
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
* These eventually lead to a terminal atom for the tape which is defined below as "tape_exit".
|
||||||
|
*
|
||||||
// Used for components with value-args (e.g., ac_format_f3_color).
|
* It behaves as one of the simplest runtime harnesses ontop of a host-enviornment's execution engine
|
||||||
// FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })
|
* to author and compose programs with. From here various conventions can be further applied.
|
||||||
// expands to:
|
* To make things easier to understand it may be better to focus on what this ABI does not have.
|
||||||
// FI_ Slice_MipsCode ac_X(args) { MipsCode ac_X[] align_(4) = { body }; return slice_from_array(MipsCode, ac_X); }
|
* It does not have have any branching within the tape but relative branches within atoms or between atoms.
|
||||||
#define MipsAtomComp_Proc_(sym, ...) { MipsCode sym [] align_(4) = __VA_ARGS__; return slice_from_array(MipsCode, sym); }
|
* Branching nearly is always downstream. Automatic stack usage is non-existent.
|
||||||
|
* Push/Pop, FIFO, or Arena/Bump data structures are used by atoms explicitly.
|
||||||
// Auto-generated component macros (<module>/gen/<dir>/<dir>.macs.h) are included manually by the unity build.
|
* In it's current form with the C11 macro DSL, the user also has fullfill manual register allocation per atom.
|
||||||
|
*
|
||||||
/* Register aliases */
|
* One of the remarkable things about utilizing this ABI is its essentially interopable with CPUs, GPUs, FPGA,
|
||||||
|
* or, basically anything from the 5th generation consoles and onward.
|
||||||
|
* The ABI directly reflects how all computational hardware must be architected in order to execute
|
||||||
|
* digital logic effectively on current era tech.
|
||||||
|
* On the PS1 we don't have access to a few features like multi-threading, speculative execution, or L3 cache;
|
||||||
|
* but, we can set the foundation for legoing whats required for eventually expanding this ABI's paradigm
|
||||||
|
* and core atoms to take those newer hardware features into account. For example, you can easily expand
|
||||||
|
* this to support wave-based execution model on a PS2 or PS3. Not having a stack or
|
||||||
|
* automatic register allocation means the user cannot ignore excessive argument shuffle across workload or
|
||||||
|
* waves and thier phases. Crossing ABI boundaries to other runtimes that do has obviouss penalties.
|
||||||
|
*
|
||||||
|
* Learning data-oriented code becomes a natural progression. Your not fighting a stack-based procedural
|
||||||
|
* paradigm that wants to argument shuffle. There is no ambiguity due to the lack of constraints, for example,
|
||||||
|
* on how the user may "call" a procedure in traditional random dispatch runtimes. The user does have to
|
||||||
|
* hammer down "rules" or patterns for massaging the compiler to dissolve those call frames; just to get
|
||||||
|
* the asesmbly into its desired form. The form is obvious, and once the user gets to author these compoonents
|
||||||
|
* it becomes a game of tetris.
|
||||||
|
*
|
||||||
|
* Another feature is this ABI is very compatible with bootstrapping and developing simple toolchains built off
|
||||||
|
* of bit-packed annotated command streams the user can directly author, maintatain, and immediately execute.
|
||||||
|
* That being like a color forth, or maybe something more familar like an immediate mode library
|
||||||
|
* for various systems such as GUIs. This can make the tetris less of a chore with some helpful policy
|
||||||
|
* generation for allocation of registers, helping to choose resuable components, designing DSL on the fly, etc.
|
||||||
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
|
* TODO(Ed): We need pretty ascii diagrams and proper guides, articles, etc.
|
||||||
|
* -----------------------------------------------------------------------------------------------------------
|
||||||
|
* For now this ideation has just started functioning. I'm abusing C11 & a lua metaprogram to help establish
|
||||||
|
* a hybrid toolchain to ideate on a traditional text-based authoring UX for this paradigm.
|
||||||
|
* If pcsx-redux provides viable hot-reload and persistent data storage beyond save-states
|
||||||
|
* (just copying ram to filesystem), I can author a color forth to mess around with.
|
||||||
|
* With either an editor in-emulator or on the actual machine itself. Assembly is tedius,
|
||||||
|
* but I think this codebase most likely has a pretty ergonomic flavor worst case...
|
||||||
|
* */
|
||||||
|
/* Register Allocation Info */
|
||||||
enum {
|
enum {
|
||||||
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
|
R_AtomJmp = R_T8 atom_reg, /* debug-visible; tape yield handshake scratch */
|
||||||
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
|
R_TapePtr = R_T9 atom_reg, /* The Instruction Stream Pointer */
|
||||||
@@ -59,28 +93,73 @@ enum {
|
|||||||
R_TScratch6 = R_T6,
|
R_TScratch6 = R_T6,
|
||||||
R_TScratch7 = R_T7,
|
R_TScratch7 = R_T7,
|
||||||
R_TScratch8 = R_T8,
|
R_TScratch8 = R_T8,
|
||||||
R_TScratch10 = R_V0,
|
R_TScratch10 = R_V0, // Tend to be used with gte DMAs
|
||||||
R_TScratch11 = R_V1,
|
R_TScratch11 = R_V1, // Tend to be used with gte DMAs
|
||||||
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
|
// Note(Ed): We can technically clobber these, but don't unless we hit a bottleneck.
|
||||||
// R_TScratch12 = R_A0,
|
// A 0-2
|
||||||
// R_TScratch13 = R_A1,
|
// S 0-7
|
||||||
// R_TScratch14 = R_A3,
|
|
||||||
// TODO(Ed): Review S0-S7, they are technically avaialble, we just have to snapshot them at the ABI boundary.
|
|
||||||
// TODO(Ed): This is technically a waste of cycles for most work? so maybe only do this for expensive atoms on-demand or atom phases.
|
|
||||||
// TODO(Ed): Sort out the other available registers... (Not sure how much is left avail)
|
|
||||||
};
|
};
|
||||||
|
|
||||||
#pragma region Tape Drive
|
typedef U2 Reg; // Register parameter used with atom or atom component procedures
|
||||||
/* ---------------------------------------------------------------------------
|
|
||||||
* TAPE DRIVE ABI & REGISTER ALIASES (the enum moved earlier; see below)
|
typedef U4 const MipsCode; // Underlying type to mips asm words.
|
||||||
* ---------------------------------------------------------------------------*/
|
typedef Slice_(MipsCode);
|
||||||
typedef Slice_(MipsAtom); typedef Slice_MipsAtom Tape;
|
|
||||||
|
typedef U4 const MipsAtom;
|
||||||
|
typedef Slice_(MipsAtom);
|
||||||
|
// Sometimes a user will define a bundle of atoms that represent a procedure of work as:
|
||||||
|
// MipsAtom* <identifier>[...];
|
||||||
|
// Unfortuantely if using slice_from_array it will make the slice's pointer: MipsAtom** so this enforce its defined as MipsAtom*
|
||||||
|
// TODO(Ed): Alternatively we can make the MipsAtom an opaque pointer to the atom... so that the blow returns 'MipsAtom'.
|
||||||
|
#define atombundle_from_array(array) (Slice_MipsAtom){.ptr=array[0],.len=Array_len(array)}
|
||||||
|
|
||||||
|
// Underlying type to an ptr to an array of mips asm words that must terminate with an ac_yield.
|
||||||
|
#define MipsAtom_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
|
// Used for atoms with value-args
|
||||||
|
// internal MipsAtom* X_proc(AtomArena_R aa, args) MipsAtom_Proc_(X, aa, { body })
|
||||||
|
// expands to:
|
||||||
|
// internal MipsAtom* X_proc(AtomArena_R aa, args) { MipsCode atom_comp_code[] align_(4) = { body }; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
// The atom name is derived by the Lua metaprogram from the preceding
|
||||||
|
// `MipsAtom* X_proc(...)` declaration (backward walk from the macro site,
|
||||||
|
// strips the `_proc` suffix).
|
||||||
|
#define MipsAtom_Proc_(aa, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; return atomarena_push(aa, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
|
||||||
|
// Used for components with no args (e.g., ac_load_tri_indices) or identifier-args (hardcoded register names).
|
||||||
|
// MipsAtomComp_(ac_X) { body }
|
||||||
|
// expands to:
|
||||||
|
// MipsCode ac_X[] align_(4) = { body };
|
||||||
|
#define MipsAtomComp_(sym) MipsCode sym [] align_(4) =
|
||||||
|
|
||||||
|
// Used for components with value-args (mandatory `ab` (atom-builder) arg).
|
||||||
|
// FI_ void ac_X(MipsAtomBuilder_R ab, args) MipsAtomComp_Proc_(ab, { body })
|
||||||
|
// expands to:
|
||||||
|
// FI_ void ac_X(MipsAtomBuilder_R ab, args) {
|
||||||
|
// MipsCode atom_comp_code[] align_(4) = { body };
|
||||||
|
// atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code));
|
||||||
|
// }
|
||||||
|
// The body must NOT include mac_yield() (the parent atom yields).
|
||||||
|
// The component name is derived by the Lua metaprogram from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration (backward walk from the macro site).
|
||||||
|
// Inline-only callers (the generated `mac_<name>` aliases) skip the `ab` arg via metaprogram filtering; escape callers (ac_<name> invoked as a function) pass a long-lived builder.
|
||||||
|
#define MipsAtomComp_Proc_(ab, ...) { MipsCode atom_comp_code[] align_(4) = __VA_ARGS__; atombuilder_push(ab, slice_from_array(MipsCode, atom_comp_code)); }
|
||||||
|
|
||||||
|
/* Line-table anchor: gcc only adds a file to the .debug_line file table when the contains line-numbered content.
|
||||||
|
Files containing only atoms and atom components.
|
||||||
|
Place `ATOM_FILE_LINE_MARKER();` once at file scope in any `.atom.c` that defines atoms.
|
||||||
|
Macro expands to a file-scope `internal U4 const` declaration keeps the file in the line table.
|
||||||
|
The constant is in `.rodata` so the linker may eliminate it.
|
||||||
|
Two-level concat + `__LINE__` suffix makes the identifier unique per call site
|
||||||
|
(identifier embeds the source line, so duplicates across `#include`d files don't collide). */
|
||||||
|
#define ATOM_FILE_DEBUGGER_LINE_MARKER(file_name) internal U4 const tmpl(atom_file_debugger_line_marker,file_name) = 0
|
||||||
|
|
||||||
|
typedef Slice_MipsAtom Tape;
|
||||||
|
|
||||||
/* The 'Exit' Atom */
|
/* The 'Exit' Atom */
|
||||||
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(rret_addr), nop };
|
atom_dbg_skip MipsAtom_(tape_exit) { jump_reg(R_RA), nop };
|
||||||
|
|
||||||
//TODO(Ed): Do we backup R_S0-7 here? Have it in a heavier tape run as a opt-in? Same with V0-1 and A0-3?
|
// TODO(Ed): When we have a substantial workload/throughput, profile each of these to see impact at ABI boundaries.
|
||||||
/* Generalized Tape Engine Runner */
|
|
||||||
|
/* Tape Runner (Default) */
|
||||||
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||||
asm_words(
|
asm_words(
|
||||||
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||||
@@ -91,29 +170,50 @@ FI_ void tape_run(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape
|
|||||||
asm_rpins, r_use(tape_ptr)
|
asm_rpins, r_use(tape_ptr)
|
||||||
asm_clobber:
|
asm_clobber:
|
||||||
rlit(R_AT),
|
rlit(R_AT),
|
||||||
rlit(R_V0), rlit(R_V1),
|
rlit(R_V0), rlit(R_V1), // We clobber these for GTE ACs (that don't expose register selection, might expose them in the future...)
|
||||||
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
||||||
clb_mem_drain
|
clb_mem_drain
|
||||||
); }
|
); }
|
||||||
|
|
||||||
|
/* Tape Runner (Static and Arg Clobbers) */
|
||||||
|
FI_ void tape_run_a02_s07(Tape tape) { register U4* tape_ptr rgcc(R_TapePtr) = u4_r(tape.ptr); asm volatile(
|
||||||
|
asm_words(
|
||||||
|
load_word( R_AtomJmp, R_TapePtr, 0) /* Bootstrap the first jump */
|
||||||
|
, add_ui_self(R_TapePtr, S_(MipsAtom)) /* Advance tape */
|
||||||
|
, call_reg( R_AtomJmp) /* jalr $t9 */
|
||||||
|
, nop /* Branch delay slot */
|
||||||
|
)
|
||||||
|
asm_rpins, r_use(tape_ptr)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1), rlit(R_A0), rlit(R_A1), rlit(R_A2),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8),
|
||||||
|
rlit(R_S0), rlit(R_S1), rlit(R_S2), rlit(R_S3), rlit(R_S4),
|
||||||
|
rlit(R_S5), rlit(R_S6), rlit(R_S7),
|
||||||
|
clb_mem_drain
|
||||||
|
); }
|
||||||
|
|
||||||
|
// Procedural authoring of tapes:
|
||||||
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(TapeBuilder) { U4 ptr; U4 capacity; U4 used; };
|
||||||
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
FI_ void tb_init(TapeBuilder* tb, FArena* arena) { tb->ptr = arena->start; tb->used = 0; }
|
||||||
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
FI_ TapeBuilder tb_make_old( FArena* arena) { return (TapeBuilder){ arena->start, 0 }; }
|
||||||
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ mem.ptr, mem.len, 0 }; }
|
FI_ TapeBuilder tb_make(Slice mem) { return (TapeBuilder){ u4_(mem.ptr), mem.len, 0 }; } /* capacity in elements (matches used units) */
|
||||||
|
|
||||||
FI_ void tb_emit(TapeBuilder* tb, MipsCode* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
FI_ void tb_emit(TapeBuilder* tb, MipsAtom* atom) { u4_r(tb->ptr)[tb->used] = u4_(atom); ++ tb->used; }
|
||||||
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
FI_ void tb_data(TapeBuilder* tb, U4 data) { u4_r(tb->ptr)[tb->used] = u4_(data); ++ tb->used; }
|
||||||
#define tb_emit_(atom) tb_emit(& tb, atom)
|
#define tb_emit_(atom) tb_emit(& tb, atom)
|
||||||
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
#define tb_data_(field, data) tb_data(& tb, u4_(data))
|
||||||
|
|
||||||
|
FI_ void tb_emit_bundle(TapeBuilder_R tb, Slice_MipsAtom atoms) { mem_copy(u4_(tb->ptr), u4_(atoms.ptr), S_slice(atoms)); tb->used += atoms.len; }
|
||||||
|
|
||||||
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
FI_ Tape tb_end (TapeBuilder* tb) { tb_emit(tb,tape_exit); return (Tape){ C_(U4*,tb->ptr), tb->used }; }
|
||||||
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
FI_ Tape tb_slice(TapeBuilder tb) { return (Tape){ C_(U4*,tb.ptr), tb.used }; }
|
||||||
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
#define tb_scope(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_emit(tb,tape_exit))
|
||||||
|
|
||||||
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
|
FI_ void tb_scope_run_end(TapeBuilder* tb) { tb_emit(tb,tape_exit); tape_run(tb_slice(tb[0])); }
|
||||||
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
|
#define tb_scope_run(tb) for(U4 tbs_once=0;tbs_once==0;++tbs_once,tb_scope_run_end(tb))
|
||||||
|
|
||||||
#pragma endregion Tape Drive
|
#pragma endregion Tape Drive
|
||||||
|
|
||||||
#pragma region Macro Mips Atom Components
|
#pragma region Macro Mips Atom Components
|
||||||
@@ -142,192 +242,146 @@ atom_dbg_skip MipsAtomComp_(ac_yield_tail) {
|
|||||||
add_ui_self(R_TapePtr, S_(MipsCode)),
|
add_ui_self(R_TapePtr, S_(MipsCode)),
|
||||||
jump_reg( R_AtomJmp), nop,
|
jump_reg( R_AtomJmp), nop,
|
||||||
};
|
};
|
||||||
|
|
||||||
enum {
|
|
||||||
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
|
||||||
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
|
||||||
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
|
||||||
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
|
||||||
#define R_PrimCursor_Code R_T7_Code
|
|
||||||
#define R_FaceCursor_Code R_T4_Code
|
|
||||||
#define R_VertBase_Code R_T5_Code
|
|
||||||
#define R_OtBase_Code R_T6_Code
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 3; Loads 3 S2 indices from the face array */
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_load_tri_indices) {
|
|
||||||
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
|
||||||
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
|
||||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 18; Translates indices to vertex addresses and pushes them to GTE */
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_gte_load_tri_verts) {
|
|
||||||
shift_lleft(R_AT, R_T0, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
|
||||||
shift_lleft(R_AT, R_T1, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY1), gte_mv_to_data_r(R_V1, C2_VZ1),
|
|
||||||
shift_lleft(R_AT, R_T2, v3s2_byteoff), add_u_self(R_AT, R_VertBase), load_word(R_V0, R_AT, O_(V3_S2,x)), load_word(R_V1, R_AT, O_(V3_S2,z)), gte_mv_to_data_r(R_V0, C2_VXY2), gte_mv_to_data_r(R_V1, C2_VZ2),
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
|
||||||
* Hardcoded for Poly_F3 (5 words). For Poly_G4, use ac_insert_ot_tag_g4. */
|
|
||||||
MipsAtomComp_(ac_insert_ot_tag_f3) {
|
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
|
||||||
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
|
|
||||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
|
||||||
load_upper_i(R_V0, (S_(Poly_F3)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (5 - 1) << 24 = 4 << 24
|
|
||||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
|
||||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
|
||||||
store_word( R_AT, R_PrimCursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
|
||||||
shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
|
||||||
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
|
||||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 11; Correctly inserts a primitive into the Ordering Table linked list.
|
|
||||||
* Hardcoded for Poly_G4 (9 words). For Poly_F3, use ac_insert_ot_tag_f3. */
|
|
||||||
MipsAtomComp_(ac_insert_ot_tag_g4) {
|
|
||||||
shift_lleft( R_T1, R_T1, S_(U4)/2), // T1 = otz * S_(U4) (otz arg is implicit R_T1)
|
|
||||||
add_u_self( R_T1, R_OtBase), // T1 = & OrderingTable[OTZ]
|
|
||||||
load_word( R_AT, R_T1, O_(PolyTag,code)), // AT = old_ot_head
|
|
||||||
load_upper_i(R_V0, (S_(Poly_G4)/S_(U4) - S_(PolyTag)/S_(U4)) << PolyTag_len_bits), // V0 = (9 - 1) << 24 = 8 << 24
|
|
||||||
mask_upper( R_AT, R_AT, S_(PolyTag_len_bits)), // Strip upper 8 bits (length from prev cell) → keep only low 24
|
|
||||||
or_u( R_AT, R_AT, R_V0), // Merge length
|
|
||||||
store_word( R_AT, R_PrimCursor, O_(PolyTag,code)), // prim->tag = packed(prim_length, old_addr)
|
|
||||||
shift_lleft( R_AT, R_PrimCursor, S_(PolyTag_len_bits)), // AT = (prim_length << 24) | old_addr
|
|
||||||
shift_lright(R_AT, R_AT, S_(PolyTag_len_bits)),
|
|
||||||
store_word( R_AT, R_T1, O_(PolyTag,code)), // OrderingTable[OTZ] = PrimCursor
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 3; Emits one (cmd|color) word to R_PrimCursor at the given
|
|
||||||
* byte offset. Internal helper used by the *_format_*_color macros. */
|
|
||||||
FI_ Slice_MipsCode ac_pack_color_word(U4 off, U4 cmd, U1 r, U1 g, U1 b)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ac_pack_color_word, {
|
|
||||||
load_upper_i(R_AT, (cmd) << 8 | (b)),
|
|
||||||
or_i_self( R_AT, ((g) << 8) | (r)),
|
|
||||||
store_word( R_AT, R_PrimCursor, (off)),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Words: 3; Emits the F3 command+color word (cmd byte | BLUE | GREEN | RED)
|
|
||||||
* Args: _r, _g, _b are 8-bit RGB byte values (not raw 16-bit fields). */
|
|
||||||
FI_ Slice_MipsCode ac_format_f3_color(U1 r, U1 g, U1 b)
|
|
||||||
atom_dbg_skip MipsAtomComp_Proc_(ac_format_f3_color, { mac_pack_color_word(O_(Poly_F3,color), gp0_cmd_poly_f3, r, g, b) })
|
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices to the F3.
|
|
||||||
* PIPELINE: post-RTPT (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen). */
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_f3) {
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_F3,p0)),
|
|
||||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_F3,p1)),
|
|
||||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_F3,p2)),
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 12; Emits the four (code|color) words of a Poly_G4.
|
|
||||||
* Args: rN,gN,bN are 8-bit RGB byte values for each of the 4 vertices. */
|
|
||||||
FI_ Slice_MipsCode ac_format_g4_color(
|
|
||||||
U1 r0, U1 g0, U1 b0,
|
|
||||||
U1 r1, U1 g1, U1 b1,
|
|
||||||
U1 r2, U1 g2, U1 b2,
|
|
||||||
U1 r3, U1 g3, U1 b3)
|
|
||||||
MipsAtomComp_Proc_(ac_format_g4_color, {
|
|
||||||
mac_pack_color_word(O_(Poly_G4,c0), gp0_cmd_poly_g4, r0,g0,b0),
|
|
||||||
mac_pack_color_word(O_(Poly_G4,c1), 0, r1,g1,b1),
|
|
||||||
mac_pack_color_word(O_(Poly_G4,c2), 0, r2,g2,b2),
|
|
||||||
mac_pack_color_word(O_(Poly_G4,c3), 0, r3,g3,b3),
|
|
||||||
})
|
|
||||||
|
|
||||||
/* Words: 3; Stores the 3 transformed (V2_S2 screen) vertices of the
|
|
||||||
* G4 triangle portion to p0/p1/p2.
|
|
||||||
* PIPELINE: post-RTPT, pre-RTPS (SXY0=v0.screen, SXY1=v1.screen, SXY2=v2.screen).
|
|
||||||
* MUST be called BEFORE V3-RTPS, otherwise SXY0/1/2 get overwritten with v3
|
|
||||||
* (RTPS writes only to SXY2, but to keep the three registers aligned with v0/v1/v2 you must store before RTPS). */
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p012) {
|
|
||||||
gte_sw(C2_SXY0, R_PrimCursor, O_(Poly_G4,p0)),
|
|
||||||
gte_sw(C2_SXY1, R_PrimCursor, O_(Poly_G4,p1)),
|
|
||||||
gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p2)),
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Words: 1; Stores the V3 screen coord to the G4's p3 slot.
|
|
||||||
* PIPELINE: post-RTPS (SXY2 holds v3.screen because RTPS writes its single-vertex result to SXY2;
|
|
||||||
* SXY0 still holds v0.screen from the earlier RTPT.
|
|
||||||
*/
|
|
||||||
atom_dbg_skip MipsAtomComp_(ac_gte_store_g4_p3) { gte_sw(C2_SXY2, R_PrimCursor, O_(Poly_G4,p3)) };
|
|
||||||
|
|
||||||
#pragma endregion Macro Atom Components
|
#pragma endregion Macro Atom Components
|
||||||
|
|
||||||
#pragma region Mips Atom Builder
|
#pragma region Atom Builder
|
||||||
// This allows for runtime procedural authoring of mips atoms.
|
// This helps with runtime procedural authoring of mips atoms.
|
||||||
|
|
||||||
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
typedef Struct_(FMipsAtom512) { U4 data[512]; U4 used; };
|
||||||
|
|
||||||
// FArena Related
|
// FArena Related
|
||||||
typedef Relative_(FArena) Struct_(MipsAtomBuilder) { U4 start; U4 capacity; U4 used; };
|
typedef Relative_(FArena) Struct_(AtomBuilder) { U4 start; U4 capacity; U4 used; };
|
||||||
// Whatever the builder is writting to should most likely coresspond
|
|
||||||
// to something that can fit within instruction cache?
|
|
||||||
|
|
||||||
FI_ void atombuilder_unroll(MipsAtomBuilder_R ab, Slice_MipsCode_R code) {
|
// Usual way to resolve an atom after the bulder is done.
|
||||||
assert(ab->capacity - ab->used - code->len);
|
#define atom_from_atombuilder(ab) C_(MipsAtom*, (ab).start)
|
||||||
mem_copy(ab->start, u4_(code->ptr), code->len);
|
|
||||||
mem_bump(ab->start, ab->capacity, & ab->used, code->len);
|
FI_ void atombuilder_push(AtomBuilder_R ab, Slice_MipsCode code) {
|
||||||
|
assert(ab->capacity - ab->used - code.len);
|
||||||
|
U4 dest = ab->start + ab->used * S_(MipsCode); U4 size = S_slice(code);
|
||||||
|
mem_copy(dest, u4_(code.ptr), size); ab->used += size;
|
||||||
}
|
}
|
||||||
#define atombuilder_unroll_mac(ab, mac) atombuilder_unroll(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
#define atombuilder_push_mac(ab, mac) atombuilder_push(ab, slice_arg_from_array(Slice_MipsCode, mac))
|
||||||
|
|
||||||
// When done authoring, utilize this to cap-off the atom
|
// When done authoring, utilize this to cap-off the atom (if not utilizing a MipsAtom_Proc).
|
||||||
FI_ void atombuilder_end(MipsAtomBuilder_R ab) {
|
FI_ void atombuilder_end(AtomBuilder_R ab) { atombuilder_push(ab, slice_from_array(MipsCode, ac_yield)); }
|
||||||
mem_copy(ab->start, u4_(ac_yield), S_(ac_yield));
|
|
||||||
mem_bump(ab->start, ab->capacity, & ab->used, S_(ac_yield));
|
|
||||||
}
|
|
||||||
|
|
||||||
#define mipsatom_from_builder(ab) (Slice_MipsCode){ab.start, ab.used}
|
|
||||||
|
|
||||||
|
FI_ void tb_emit_atombuilder(TapeBuilder_R tb, AtomBuilder_R ab) { tb_emit(tb, atom_from_atombuilder(ab[0])); }
|
||||||
#pragma endregion Mips Atom Builder
|
#pragma endregion Mips Atom Builder
|
||||||
|
|
||||||
|
#pragma region Atom Arena
|
||||||
|
// Just a dedicated FArena that is meant to mem_copy and return atom definitions made with MipsAtom_Proc_
|
||||||
|
|
||||||
|
typedef Relative_(FArena) Struct_(AtomArena) { U4 start; U4 capacity; U4 used; };
|
||||||
|
|
||||||
|
#define atomarena_unused_start(ab) ((ab).start + (ab).used)
|
||||||
|
FI_ void atomarena_init(AtomArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
|
arena->start = u4_(mem.ptr);
|
||||||
|
arena->capacity = mem.len;
|
||||||
|
arena->used = 0;
|
||||||
|
}
|
||||||
|
FI_ AtomArena atomarena_make(Slice mem) { AtomArena a; atomarena_init(& a, mem); return a; }
|
||||||
|
FI_ MipsAtom* atomarena_push(AtomArena_R aa, Slice_MipsCode code) {
|
||||||
|
assert(aa->capacity - aa->used - code.len);
|
||||||
|
U4 dest = atomarena_unused_start(aa[0]); U4 size = S_slice(code);
|
||||||
|
mem_copy(dest, u4_(code.ptr), size); aa->used += size;
|
||||||
|
return C_(MipsAtom*, dest);
|
||||||
|
}
|
||||||
|
FI_ void atomarena_reset(AtomArena_R aa) { aa->used = 0; }
|
||||||
|
#pragma region Atom Arena
|
||||||
|
|
||||||
|
#pragma region RegFile (Register File Allocator)
|
||||||
|
// A specialized allocator utilized to help the user track which registers are bound to values
|
||||||
|
// that must be preserved for the arena's bounds.
|
||||||
|
// TODO(Ed): Technically we can do this at comp-time with the metaprogram, but we may have namespace conflicts.
|
||||||
|
// Unless we follow a convention for #define <Scope_Prefix> or something per register allocation boundary.
|
||||||
|
|
||||||
|
/* ABI + tape reserves that are never handed out by alloc. */
|
||||||
|
U4 const regfile_abi_mask =
|
||||||
|
(1u << R_0) | (1u << R_AT) |
|
||||||
|
(1u << R_K0) | (1u << R_K1) |
|
||||||
|
(1u << R_GP) | (1u << R_SP) |
|
||||||
|
(1u << R_FP) | (1u << R_RA) |
|
||||||
|
(1u << R_T8) | (1u << R_T9); /* AtomJmp + TapePtr */
|
||||||
|
|
||||||
|
typedef Struct_(RegFile) {
|
||||||
|
A2_U2 GPR;
|
||||||
|
A2_U2 GTE;
|
||||||
|
};
|
||||||
|
#define regfile(pin_mask) {.GPR={u4_lo(pin_mask), u4_hi(pin_mask)} }
|
||||||
|
FI_ void regfile_init(RegFile_R rf) {
|
||||||
|
/* pack the 32-bit ABI mask into the two U2s */
|
||||||
|
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||||
|
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||||
|
rf->GTE[0] = rf->GTE[1] = 0;
|
||||||
|
}
|
||||||
|
FI_ RegFile regfile_make(void) { RegFile rf; regfile_init(& rf); return rf; }
|
||||||
|
|
||||||
|
typedef Struct_(RegFile_RInfo) {
|
||||||
|
U2_R section;
|
||||||
|
U2 mask;
|
||||||
|
B2 occupied;
|
||||||
|
};
|
||||||
|
FI_ RegFile_RInfo regfile_rinfo(A2_U2 file, Reg r_id) {
|
||||||
|
U2 s_id = r_id >> 4;
|
||||||
|
U2_R section = & file[s_id];
|
||||||
|
U2 mask = u2_(1u << (r_id & 15));
|
||||||
|
B2 occupied = (section[0] & mask) != 0;
|
||||||
|
return (RegFile_RInfo){section, mask, occupied};
|
||||||
|
}
|
||||||
|
FI_ Reg regfile__alloc_helper(A2_U2 file, Reg r_id) {
|
||||||
|
Reg result = 0; RegFile_RInfo info = regfile_rinfo(file, r_id);
|
||||||
|
if (info.occupied == false) {
|
||||||
|
info.section[0] |= info.mask;
|
||||||
|
result = r_id;
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
I_ Reg regfile_alloc(RegFile_R rf) {
|
||||||
|
U2 allocated = 0;
|
||||||
|
for index_iter(Reg, r_id, R_T0, <=, R_T7) {
|
||||||
|
allocated = regfile__alloc_helper(rf->GPR, r_id); Jmp_nZero_(allocated,resolved);
|
||||||
|
}
|
||||||
|
allocated = regfile__alloc_helper(rf->GPR, R_V0); Jmp_nZero_(allocated,resolved);
|
||||||
|
allocated = regfile__alloc_helper(rf->GPR, R_V1);
|
||||||
|
assert(allocated != 0);
|
||||||
|
resolved: return allocated;
|
||||||
|
}
|
||||||
|
FI_ Reg regfile_pin(RegFile_R rf, Reg r_id) {
|
||||||
|
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||||
|
assert(info.occupied == false);
|
||||||
|
info.section[0] |= info.mask;
|
||||||
|
return r_id;
|
||||||
|
}
|
||||||
|
FI_ void regfile_pin_mask(RegFile_R rf, U4 mask) {
|
||||||
|
B4 occupied = u4_r(rf->GPR)[0] & mask;
|
||||||
|
assert(occupied == false);
|
||||||
|
u4_r(rf->GPR)[0] |= mask;
|
||||||
|
}
|
||||||
|
FI_ void regfile_free_mask(RegFile_R rf, U4 mask) {
|
||||||
|
if (regfile_abi_mask & mask) return;
|
||||||
|
u4_r(rf->GPR)[0] &= ~mask;
|
||||||
|
}
|
||||||
|
FI_ void regfile_free_reg(RegFile_R rf, Reg r_id) {
|
||||||
|
/* never free the ABI set */
|
||||||
|
if (regfile_abi_mask & (1u << r_id)) return;
|
||||||
|
RegFile_RInfo info = regfile_rinfo(rf->GPR, r_id);
|
||||||
|
info.section[0] &= ~info.mask;
|
||||||
|
}
|
||||||
|
FI_ void regfile_reset(RegFile_R rf) {
|
||||||
|
rf->GPR[0] = u4_lo(regfile_abi_mask);
|
||||||
|
rf->GPR[1] = u4_hi(regfile_abi_mask);
|
||||||
|
}
|
||||||
|
FI_ void regfile_reset_mask(RegFile_R rf, U4 mask) {
|
||||||
|
rf->GPR[0] = u4_lo(mask);
|
||||||
|
rf->GPR[1] = u4_hi(mask);
|
||||||
|
}
|
||||||
|
#pragma endregion RegFileArena (Register File Allocator)
|
||||||
|
|
||||||
|
#pragma region Mips Atom Procs
|
||||||
|
|
||||||
|
#pragma endregion Mips Atom Procs
|
||||||
|
|
||||||
#pragma region Baked Mips Atoms
|
#pragma region Baked Mips Atoms
|
||||||
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
// These atoms are resolved at compile time and are (usually) statically linked readonly data.
|
||||||
|
|
||||||
enum {
|
|
||||||
bios_flushcache = 0x44,
|
|
||||||
bios_table_addr = 0xA0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
|
||||||
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
|
||||||
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
|
||||||
* 2. $a0 = bios_flushcache (arg0)
|
|
||||||
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
|
||||||
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
|
||||||
* nop ; branch delay slot
|
|
||||||
* 5. lw $ra, 4($sp); jr $ra ; restore & return
|
|
||||||
* 6. sp += 8
|
|
||||||
*/
|
|
||||||
internal MipsAtom_(mips_flush_icache) {
|
|
||||||
add_ui(rstack_ptr, rstack_ptr, -MipsStackAlignment), // sp -= 8
|
|
||||||
store_word(rret_addr, rstack_ptr, S_(U4)), // sw $ra, 4($sp)
|
|
||||||
add_ui(rret_0, rdiscard, bios_flushcache), // addiu $a0, $0, 0x44
|
|
||||||
add_ui(rtmp_0, rdiscard, bios_table_addr), // addiu $t0, $0, 0xA0
|
|
||||||
jump_link(rtmp_0, rret_addr), nop, // jalr $t0, $ra, BD slot
|
|
||||||
load_word(rret_addr, rstack_ptr, S_(U4)), // lw $ra, 4($sp)
|
|
||||||
jump_reg(rret_addr), // jr $ra
|
|
||||||
add_ui(rstack_ptr, rstack_ptr, MipsStackAlignment), // sp += 8 (BD)
|
|
||||||
mac_yield(),
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef Struct_(Binds_SetGteWorld) {
|
|
||||||
M3_S2* transform;
|
|
||||||
};
|
|
||||||
internal MipsAtom_(set_gte_world) atom_info(
|
|
||||||
atom_bind(Binds_SetGteWorld)
|
|
||||||
, atom_reads(R_TapePtr)
|
|
||||||
){
|
|
||||||
/* Pop matrix address from tape into R_T3 ($11) */
|
|
||||||
load_word(R_T3, R_TapePtr, O_(Binds_SetGteWorld,transform)),
|
|
||||||
add_ui_self( R_TapePtr, S_(Binds_SetGteWorld)),
|
|
||||||
/* Load 3x3 Rotation + 3x1 Translation from R_T3 into GTE CONTROL Regs (ctc2) */
|
|
||||||
load_word(R_T0, R_T3, 0), load_word(R_T1, R_T3, 4),
|
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT11), gte_mv_to_ctrl_r(R_T1, gte_cr_RT12),
|
|
||||||
load_word(R_T0, R_T3, 8), load_word(R_T1, R_T3, 12), load_word(R_T2, R_T3, 16),
|
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_RT13), gte_mv_to_ctrl_r(R_T1, gte_cr_RT21), gte_mv_to_ctrl_r(R_T2, gte_cr_RT22),
|
|
||||||
load_word(R_T0, R_T3, 20), load_word(R_T1, R_T3, 24), load_word(R_T2, R_T3, 28),
|
|
||||||
gte_mv_to_ctrl_r(R_T0, gte_cr_TRX), gte_mv_to_ctrl_r(R_T1, gte_cr_TRY), gte_mv_to_ctrl_r(R_T2, gte_cr_TRZ),
|
|
||||||
mac_yield()
|
|
||||||
};
|
|
||||||
|
|
||||||
#pragma endregion Baked Mips Atoms
|
#pragma endregion Baked Mips Atoms
|
||||||
|
|||||||
@@ -0,0 +1,55 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "math.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(math_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Component)
|
||||||
|
|
||||||
|
// FI_ Slice_MipsCode ac_load_imm
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v2s2(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_half( rs_x, r_base, offset + O_(V3_S2,x)),
|
||||||
|
load_half( rs_y, r_base, offset + O_(V3_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v2s2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_v3s4(AtomBuilder_R ab, U4 rs_x, U4 rs_y, U4 rs_z, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_word( rs_x, r_base, offset + O_(V3_S4,x)),
|
||||||
|
load_word( rs_y, r_base, offset + O_(V3_S4,y)),
|
||||||
|
load_word( rs_z, r_base, offset + O_(V3_S4,z)),
|
||||||
|
})
|
||||||
|
// TODO(Ed): we could generate these mappings properly..
|
||||||
|
#define ac_load_p3s4 ac_load_v3s4
|
||||||
|
#define mac_load_p3s4 mac_load_v3s4
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_v3s4(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_z, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_word(rt_x, base, offset + O_(V3_S4,x)),
|
||||||
|
store_word(rt_y, base, offset + O_(V3_S4,y)),
|
||||||
|
store_word(rt_z, base, offset + O_(V3_S4,z)),
|
||||||
|
})
|
||||||
|
// TODO(Ed): we could generate these mappings properly..
|
||||||
|
#define ac_store_p3s4 ac_store_v3s4
|
||||||
|
#define mac_store_p3s4 mac_store_v3s4
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_sub_v3s4(AtomBuilder_R ab, U4 rds_x, U4 rds_y, U4 rds_z, U4 rt_x, U4 rt_y, U4 rt_z) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
sub_s(rds_x, rds_x, rt_x),
|
||||||
|
sub_s(rds_y, rds_y, rt_y),
|
||||||
|
sub_s(rds_z, rds_z, rt_z),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_store_rects2(AtomBuilder_R ab, U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
||||||
|
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
||||||
|
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
||||||
|
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Component)
|
||||||
+62
-11
@@ -7,11 +7,24 @@
|
|||||||
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
#define max(A, B) (((A) > (B)) ? (A) : (B))
|
||||||
#define clamp_bot(X, B) max(X, B)
|
#define clamp_bot(X, B) max(X, B)
|
||||||
|
|
||||||
|
/* Convention
|
||||||
|
<Type> ## <Width> _ <Component Type> ## <Component Width>
|
||||||
|
For types with compound data (Ex: Rotation Matrix & Translation):
|
||||||
|
<TypeA> ## <TypeB> ## <Width> _ <ComponentTypeA> ## <ComponentWidthA> ## <ComponentTypeB> ## <ComponentWidthB>
|
||||||
|
|
||||||
|
A: Array
|
||||||
|
V: Vector
|
||||||
|
R: Range
|
||||||
|
M: Matrix
|
||||||
|
T: Translation
|
||||||
|
*/
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
v3s2_byteoff = 3, // log2(8), used with shift_left_logical op for index via byte offset.
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Array_(U1, 2);
|
typedef Array_(U1, 2);
|
||||||
|
typedef Array_(U2, 2);
|
||||||
typedef Array_(U4, 2);
|
typedef Array_(U4, 2);
|
||||||
typedef Array_(S2, 2);
|
typedef Array_(S2, 2);
|
||||||
typedef Array_(S2, 3);
|
typedef Array_(S2, 3);
|
||||||
@@ -26,23 +39,43 @@ typedef Struct_(Extent2_S4) { S4 width; S4 height; };
|
|||||||
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
typedef Struct_(V2_U1) { U1 x; U1 y; };
|
||||||
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
typedef Struct_(V2_S2) { S2 x; S2 y; };
|
||||||
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
typedef Struct_(V2_S4) { S4 x; S4 y; };
|
||||||
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; };
|
typedef Struct_(V3_S2) { S2 x; S2 y; S2 z; S2 pad; }; // PSY-Q: SVECTOR
|
||||||
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; };
|
typedef Struct_(V3_S4) { S4 x; S4 y; S4 z; S4 pad; }; // PSY-Q: VECTOR. RGA(Lengyel): Euclidean vector or direction. A zero-weight RGA point is stored as a V3_S4 with the implicit weight dropped.
|
||||||
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
typedef Struct_(V4_S2) { S2 x; S2 y; S2 z; S2 w; };
|
||||||
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
typedef Struct_(V4_S4) { S4 x; S4 y; S4 z; S4 w; };
|
||||||
|
|
||||||
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; };
|
// typedef Struct_(P3_S4) { S4 x; S4 y; S4 z; S4 w1; }; // RGA(Lengyel): Affine point with implicit weight one. Storage alias of V3_S4. Use P3_S4 when the value is a point.
|
||||||
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; };
|
typedef V3_S4 P3_S4;
|
||||||
|
|
||||||
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
typedef Struct_(R1_U2) { U2 p0; U2 p1; };
|
||||||
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
typedef Struct_(R1_S2) { S2 p0; S2 p1; };
|
||||||
|
|
||||||
typedef Struct_(M3_S2) { A3x3_S2 m; A3_S4 t; };
|
typedef Struct_(R2_S2) { V2_S2 p0; V2_S2 p1; }; // Range-2 Signed 2-Byte (16-bit)
|
||||||
|
typedef Struct_(R2_S4) { V2_S4 p0; V2_S4 p1; }; // Range-2 Signed 4-Byte (32-bit)
|
||||||
|
|
||||||
|
typedef Struct_(Rect_S2) { S2 x; S2 y; S2 width; S2 height; };
|
||||||
|
typedef Struct_(Rect_S4) { S4 x; S4 y; S4 width; S4 height; };
|
||||||
|
|
||||||
|
typedef Struct_(MT3_S2S4) { A3x3_S2 m; A3_S4 t; }; // PSY-Q: MATRIX. RGA(Lengyel): Matrix expansion of a rigid transformation. GTE utilizes this representation; corresponding motor not constructed here.
|
||||||
|
|
||||||
|
/* RGA(Lengyel) reserved names (deferred):
|
||||||
|
* P4_S4 - future flat point with explicit weight (Lengyel/TML FlatPoint3D analog).
|
||||||
|
* B3_S4 - future 3D bivector (callers store a Complement(Wedge(...)) as a V3_S4).
|
||||||
|
* Mo8_S4 - future motor. Not introduced until a course operation actually needs composition, interpolation, or inversion. */
|
||||||
|
|
||||||
|
typedef Array_(V2_U1, 2);
|
||||||
typedef Array_(V2_S2, 2);
|
typedef Array_(V2_S2, 2);
|
||||||
typedef Array_(V2_S2, 3);
|
typedef Array_(V2_S2, 3);
|
||||||
typedef Array_(V2_S2, 4);
|
typedef Array_(V2_S2, 4);
|
||||||
|
|
||||||
|
#define r1u2(p0,p1) (R1_U2){p0,p1}
|
||||||
|
|
||||||
|
enum {
|
||||||
|
fp_one = (1 << 12),
|
||||||
|
};
|
||||||
|
|
||||||
|
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||||
|
|
||||||
#define v2s2(x,y) (V2_S2){x,y}
|
#define v2s2(x,y) (V2_S2){x,y}
|
||||||
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
#define v3s2(x,y,z) (V3_S2){x,y,z,0}
|
||||||
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
#define v3s4(x,y,z) (V3_S4){x,y,z,0}
|
||||||
@@ -61,10 +94,28 @@ FI_ void add_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
|||||||
(out_a[0])[2] += b[2] >> 1;
|
(out_a[0])[2] += b[2] >> 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
FI_ void add_v3s4(V3_S4_R out_a, V3_S4 b) {
|
FI_ void sub_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||||
add_a3s4(pcast(A3_S4_R, out_a), pcast(A3_S4, b));
|
(out_a[0])[0] -= b[0];
|
||||||
|
(out_a[0])[1] -= b[1];
|
||||||
|
(out_a[0])[2] -= b[2];
|
||||||
}
|
}
|
||||||
|
|
||||||
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) {
|
FI_ void sub_a3s4_fp(A3_S4_R out_a, A3_S4 b) {
|
||||||
add_a3s4_fp(pcast(A3_S4_R, out_a), pcast(A3_S4, b));
|
(out_a[0])[0] -= b[0] >> 1;
|
||||||
|
(out_a[0])[1] -= b[1] >> 1;
|
||||||
|
(out_a[0])[2] -= b[2] >> 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
FI_ void mul_a3s4(A3_S4_R out_a, A3_S4 b) {
|
||||||
|
(out_a[0])[0] *= b[0];
|
||||||
|
(out_a[0])[1] *= b[1];
|
||||||
|
(out_a[0])[2] *= b[2];
|
||||||
|
}
|
||||||
|
|
||||||
|
FI_ void add_v3s4 (V3_S4_R out_a, V3_S4 b) { add_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
FI_ void add_v3s4_fp(V3_S4_R out_a, V3_S4 b) { add_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
|
||||||
|
FI_ void sub_v3s4 (V3_S4_R out_a, V3_S4 b) { sub_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
FI_ void sub_v3s4_fp(V3_S4_R out_a, V3_S4 b) { sub_a3s4_fp(C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
|
||||||
|
FI_ void mul_v3s4 (V3_S4_R out_a, V3_S4 b) { mul_a3s4 (C_ptr(A3_S4_R, out_a), C_ptr(A3_S4, b)); }
|
||||||
|
|||||||
+22
-13
@@ -18,7 +18,7 @@ I_ U4 align_pow2(U4 x, U4 b) {
|
|||||||
|
|
||||||
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
#define align_struct(type_width) ((U4)(((type_width) + 3) & ~3))
|
||||||
|
|
||||||
FI_ void mem_bump(U4 start, U4 cap, U4*R_ used, U4 amount) {
|
FI_ void mem_bump(U4 cap, U4*R_ used, U4 amount) {
|
||||||
assert(amount <= (cap - used[0]));
|
assert(amount <= (cap - used[0]));
|
||||||
used[0] += amount;
|
used[0] += amount;
|
||||||
}
|
}
|
||||||
@@ -58,13 +58,13 @@ typedef Struct_(Str8) { UTF8* ptr; U4 len; };
|
|||||||
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
typedef Struct_(Slice_Str8) { Str8* ptr; U4 len; };
|
||||||
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
#define slit(string_literal) (Str8){ (UTF8*) string_literal, S_(string_literal) - 1 }
|
||||||
|
|
||||||
typedef Struct_(Slice) { U4 ptr, len; }; // Untyped Slice
|
typedef Struct_(Slice) { B1* ptr; U4 len; }; // Untyped Slice (byte-addressable; .len in elements)
|
||||||
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){ptr, len}; }
|
FI_ Slice slice_ut_(U4 ptr, U4 len) { return (Slice){(B1*)ptr, len}; }
|
||||||
|
|
||||||
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
#define Slice_(type) Struct_(tmpl(Slice,type)) { type* ptr; U4 len; }
|
||||||
typedef Slice_(B1);
|
typedef Slice_(B1);
|
||||||
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
#define slice_assert(s) do { assert((s).ptr != 0); assert((s).len > 0); } while(0)
|
||||||
#define slice_end(slice) ((slice).ptr + (slice).len)
|
#define slice_end(slice) ((slice).ptr + S_slice(slice) / S_(B1)) /* byte-ptr arithmetic; .len is in elements per slice convention */
|
||||||
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
#define S_slice(s) ((s).len * S_((s).ptr[0]))
|
||||||
|
|
||||||
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
#define slice_ut(ptr,len) slice_ut_(u4_(ptr), u4_(len))
|
||||||
@@ -72,23 +72,30 @@ typedef Slice_(B1);
|
|||||||
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
#define slice_to_ut(s) slice_ut_(u4_((s).ptr), S_slice(s))
|
||||||
|
|
||||||
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
#define slice_iter(container, iter) (T_((container).ptr) iter = (container).ptr; iter != slice_end(container); ++ iter)
|
||||||
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = array_decl(type,__VA_ARGS__), .len = array_len( array_decl(type,__VA_ARGS__)) }
|
#define slice_arg_from_array(type, ...) & (tmpl(Slice,type)) { .ptr = Array_decl(type,__VA_ARGS__), .len = Array_len( Array_decl(type,__VA_ARGS__)) }
|
||||||
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = S_(array) }
|
#define slice_from_array(type, array) (tmpl(Slice,type)) { .ptr = array, .len = Array_len(array) }
|
||||||
|
|
||||||
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(s.ptr, s.len); }
|
FI_ void slice_zero_(Slice s) { slice_assert(s); mem_zero(u4_(s.ptr), s.len); }
|
||||||
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
#define slice_zero(s) slice_zero_(slice_to_ut(s))
|
||||||
|
|
||||||
FI_ void slice_copy_(Slice dest, Slice src) {
|
FI_ void slice_copy_(Slice dest, Slice src) {
|
||||||
assert(dest.len >= src.len);
|
assert(S_slice(dest) >= S_slice(src));
|
||||||
slice_assert(dest);
|
slice_assert(dest);
|
||||||
slice_assert(src);
|
slice_assert(src);
|
||||||
mem_copy(dest.ptr, src.ptr, src.len);
|
mem_copy(u4_(dest.ptr), u4_(src.ptr), S_slice(src));
|
||||||
}
|
}
|
||||||
#define slice_copy(dest, src) do { \
|
#define slice_copy(dest, src) do { \
|
||||||
static_assert(T_same(dest, src)); \
|
static_assert(T_same(dest, src)); \
|
||||||
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
slice_copy_(slice_to_ut(dest), slice_to_ut(src)); \
|
||||||
} while(0)
|
} while(0)
|
||||||
|
|
||||||
|
FI_ Slice slice_bump(U4_R used, U4 start, U4 len, U4 amount) {
|
||||||
|
assert(len - used[0] - amount);
|
||||||
|
U4 ptr = start + used[0]; used[0] += amount;
|
||||||
|
return slice_ut(ptr, amount);
|
||||||
|
}
|
||||||
|
|
||||||
|
typedef Slice_(U1);
|
||||||
typedef Slice_(U4);
|
typedef Slice_(U4);
|
||||||
|
|
||||||
#pragma endregion Slice
|
#pragma endregion Slice
|
||||||
@@ -98,18 +105,19 @@ typedef Slice_(U4);
|
|||||||
typedef Opt_(farena) { U4 alignment, type_width; };
|
typedef Opt_(farena) { U4 alignment, type_width; };
|
||||||
typedef Struct_(FArena) { U4 start, capacity, used; };
|
typedef Struct_(FArena) { U4 start, capacity, used; };
|
||||||
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
FI_ void farena_init(FArena_R arena, Slice mem) { assert(arena != nullptr);
|
||||||
arena->start = mem.ptr;
|
arena->start = u4_(mem.ptr);
|
||||||
arena->capacity = mem.len;
|
arena->capacity = mem.len;
|
||||||
arena->used = 0;
|
arena->used = 0;
|
||||||
}
|
}
|
||||||
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
FI_ FArena farena_make(Slice mem) { FArena a; farena_init(& a, mem); return a; }
|
||||||
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
FI_ Slice farena_bump(FArena_R a, U4 amount) { return slice_bump(& a->used, a->start, a->capacity, amount); }
|
||||||
|
I_ Slice farena_push(FArena_R arena, U4 amount, Opt_farena o) {
|
||||||
if (amount == 0) { return (Slice){}; }
|
if (amount == 0) { return (Slice){}; }
|
||||||
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
U4 desired = amount * (o.type_width == 0 ? 1 : o.type_width);
|
||||||
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
U4 to_commit = align_pow2(desired, o.alignment ? o.alignment : MEM_ALIGNMENT_DEFAULT);
|
||||||
U4 ptr = arena->start + arena->used;
|
U4 ptr = arena->start + arena->used;
|
||||||
mem_bump(arena->start, arena->capacity, & arena->used, to_commit);
|
mem_bump(arena->capacity, & arena->used, to_commit);
|
||||||
return (Slice){ ptr, to_commit };
|
return (Slice){ (B1*)ptr, to_commit };
|
||||||
}
|
}
|
||||||
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
FI_ void farena_reset (FArena_R arena) { arena->used = 0; }
|
||||||
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
||||||
@@ -117,6 +125,7 @@ FI_ void farena_rewind(FArena_R arena, U4 save_point) {
|
|||||||
arena->used -= save_point - arena->start;
|
arena->used -= save_point - arena->start;
|
||||||
}
|
}
|
||||||
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
FI_ U4 farena_save(FArena arena) { return arena.used; }
|
||||||
|
FI_ U4 farena_unused_start(FArena arena) { return arena.start + arena.used; }
|
||||||
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
#define farena_push_(arena, amount, ...) farena_push((arena), (amount), opt_(farena, __VA_ARGS__))
|
||||||
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
#define farena_push_type(arena, type, ...) C_(type*, farena_push((arena), 1, opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr)
|
||||||
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
#define farena_push_array(arena, type, amount, ...) (tmpl(Slice,type)){ C_(type*, farena_push((arena), (amount), opt_(farena, .type_width=S_(type), __VA_ARGS__)).ptr), (amount) }
|
||||||
|
|||||||
@@ -0,0 +1,45 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "bios.h"
|
||||||
|
# include "mips.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(mips_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_load_word_imm(AtomBuilder_R ab, Reg dst, U4 imm)
|
||||||
|
atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_upper_i(dst, u4_hi(imm)),
|
||||||
|
or_i_self( dst, u4_lo(imm)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
/* Flushes the Instruction Cache (PSX A-function 0x44 via BIOS stub at 0xA0).
|
||||||
|
* Sequence (per MIPS ABI; arguments in arg registers, RA pushed to stack):
|
||||||
|
* 1. sp -= 8; sw $ra, 4($sp) ; save RA
|
||||||
|
* 2. $a0 = bios_flushcache (arg0)
|
||||||
|
* 3. $t0 = bios_table_addr ; t0 = &BIOS A-function table
|
||||||
|
* 4. jalr $t0, $ra ; call BIOS(flushcache)
|
||||||
|
* nop ; branch delay slot
|
||||||
|
* 5. lw $ra, 4($sp); jr $ra ; restore & return
|
||||||
|
* 6. sp += 8
|
||||||
|
*/
|
||||||
|
internal MipsAtom_(mips_flush_icache) {
|
||||||
|
add_ui(R_SP, R_SP, -MipsStackAlignment), // sp -= 8
|
||||||
|
store_word(R_RA, R_SP, S_(U4)), // sw $ra, 4($sp)
|
||||||
|
add_ui(R_V0, R_0, bios_flushcache), // addiu $a0, $0, 0x44
|
||||||
|
add_ui(R_T0, R_0, bios_table_addr), // addiu $t0, $0, 0xA0
|
||||||
|
jump_link(R_T0, R_RA), nop, // jalr $t0, $ra, BD slot
|
||||||
|
load_word(R_RA, R_SP, S_(U4)), // lw $ra, 4($sp)
|
||||||
|
jump_reg(R_RA), // jr $ra
|
||||||
|
add_ui(R_SP, R_SP, MipsStackAlignment), // sp += 8 (BD)
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
+58
-49
@@ -136,31 +136,31 @@ enum {
|
|||||||
|
|
||||||
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
/* Semantic Aliases for MIPS Registers (O32 ABI) */
|
||||||
|
|
||||||
, rdiscard = R_0 /* Hardwired to 0 */
|
// , rdiscard = R_0 /* Hardwired to 0 */
|
||||||
, rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
// , rasm_tmp = R_AT /* Assembler temporary (destroyed by some assembler pseudoinstructions!) */
|
||||||
, rret_0 = R_V0 /* Function return value */
|
// , rret_0 = R_V0 /* Function return value */
|
||||||
, rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
// , rret_1 = R_V1 /* Second return value (e.g., 64-bit) */
|
||||||
, rarg_0 = R_A0 /* First function argument */
|
// , rarg_0 = R_A0 /* First function argument */
|
||||||
, rarg_1 = R_A1 /* Second function argument */
|
// , rarg_1 = R_A1 /* Second function argument */
|
||||||
, rarg_2 = R_A2 /* Third function argument */
|
// , rarg_2 = R_A2 /* Third function argument */
|
||||||
, rarg_3 = R_A3 /* Fourth function argument */
|
// , rarg_3 = R_A3 /* Fourth function argument */
|
||||||
, rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
// , rtmp_0 = R_T0 /* Temporary (Caller saved) */
|
||||||
, rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
// , rtmp_1 = R_T1 /* Temporary (Caller saved) */
|
||||||
, rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
// , rtmp_2 = R_T2 /* Temporary (Caller saved) */
|
||||||
, rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
// , rtmp_3 = R_T3 /* Temporary (Caller saved) */
|
||||||
, rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
// , rtmp_4 = R_T4 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
, rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
// , rtmp_9 = R_T9 /* Temporary (Caller saved) — common GTE base pointer */
|
||||||
, rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
// , rstatic_0 = R_S0 /* Static (Callee saved, preserved across calls) */
|
||||||
, rstatic_1 = R_S1
|
// , rstatic_1 = R_S1
|
||||||
, rstatic_2 = R_S2
|
// , rstatic_2 = R_S2
|
||||||
, rstatic_3 = R_S3
|
// , rstatic_3 = R_S3
|
||||||
, rstatic_4 = R_S4
|
// , rstatic_4 = R_S4
|
||||||
, rstatic_5 = R_S5
|
// , rstatic_5 = R_S5
|
||||||
, rstatic_6 = R_S6
|
// , rstatic_6 = R_S6
|
||||||
, rstatic_7 = R_S7
|
// , rstatic_7 = R_S7
|
||||||
, rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
// , rsaved_0 = R_S0 /* Alias for rstatic_0 (alternate vocabulary) */
|
||||||
, rstack_ptr = R_SP /* Stack Pointer */
|
// , rstack_ptr = R_SP /* Stack Pointer */
|
||||||
, rret_addr = R_RA /* Return Address (populated by JAL) */
|
// , rret_addr = R_RA /* Return Address (populated by JAL) */
|
||||||
|
|
||||||
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
/* --- MIPS CPU Opcodes (Bits 31-26) --- */
|
||||||
|
|
||||||
@@ -259,22 +259,22 @@ enum { _BitOffsets = 0
|
|||||||
, SHAMT_SHIFT = 6 /* Shift Amount */
|
, SHAMT_SHIFT = 6 /* Shift Amount */
|
||||||
, FC_SHIFT = 0
|
, FC_SHIFT = 0
|
||||||
|
|
||||||
/* Bit Masks to prevent overflow into adjacent fields */
|
/* IMM_MASK is the 16-bit two's-complement truncation for the immediate field.
|
||||||
|
* It is NOT a range guard — it is load-bearing for negative branch offsets
|
||||||
|
* (the metaprogram emits raw signed offsets; the mask truncates them to the
|
||||||
|
* 16-bit representation the hardware expects). The static analysis
|
||||||
|
* `immediate_field_width` check validates ranges at build time. */
|
||||||
|
|
||||||
, OPCODE_MASK = 0x3F
|
|
||||||
, REG_MASK = 0x1F
|
|
||||||
, SHAMT_MASK = 0x1F /* Shift Amount */
|
|
||||||
, FC_MASK = 0x3F
|
|
||||||
, IMM_MASK = 0xFFFF
|
, IMM_MASK = 0xFFFF
|
||||||
};
|
};
|
||||||
|
|
||||||
#define enc_op(op) (((op) & OPCODE_MASK) << OPCODE_SHIFT)
|
#define enc_op(op) ((op) << OPCODE_SHIFT)
|
||||||
#define enc_rs(rs) (((rs) & REG_MASK) << RS_SHIFT)
|
#define enc_rs(rs) ((rs) << RS_SHIFT)
|
||||||
#define enc_rt(rt) (((rt) & REG_MASK) << RT_SHIFT)
|
#define enc_rt(rt) ((rt) << RT_SHIFT)
|
||||||
#define enc_rd(rd) (((rd) & REG_MASK) << RD_SHIFT)
|
#define enc_rd(rd) ((rd) << RD_SHIFT)
|
||||||
#define enc_shamt(shamt) (((shamt) & SHAMT_MASK) << SHAMT_SHIFT)
|
#define enc_shamt(shamt) ((shamt) << SHAMT_SHIFT)
|
||||||
#define enc_fc(fc) (((fc) & FC_MASK) << FC_SHIFT)
|
#define enc_fc(fc) ((fc) << FC_SHIFT)
|
||||||
#define enc_imm(imm) (((imm) & IMM_MASK))
|
#define enc_imm(imm) ((imm) & IMM_MASK)
|
||||||
|
|
||||||
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
/* MIPS R-Type Instruction Format (Register-to-Register) */
|
||||||
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
#define enc_r(op, rs, rt, rd, shamt, fc) (enc_op(op) | enc_rs(rs) | enc_rt(rt) | enc_rd(rd) | enc_shamt(shamt) | enc_fc(fc))
|
||||||
@@ -318,7 +318,10 @@ enum { _BitOffsets = 0
|
|||||||
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
#define load_half(rt, base, off) enc_i(op_lh, (base), (rt), (off))
|
||||||
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
#define load_byte_u(rt, base, off) enc_i(op_lbu, (base), (rt), (off))
|
||||||
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
#define load_half_u(rt, base, off) enc_i(op_lhu, (base), (rt), (off))
|
||||||
|
#define LdSlot_
|
||||||
|
|
||||||
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
#define store_word(rt, base, off) enc_i(op_sw, (base), (rt), (off))
|
||||||
|
|
||||||
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
#define add_ui(rt, rs, imm) enc_i(op_addiu, (rs), (rt), (imm))
|
||||||
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
#define and_i(rt, rs, imm) enc_i(op_andi, (rs), (rt), (imm))
|
||||||
// #define and_si and_i
|
// #define and_si and_i
|
||||||
@@ -348,6 +351,12 @@ enum { _BitOffsets = 0
|
|||||||
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
#define shift_lright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_srl)
|
||||||
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
#define shift_aright(rd, rt, shamt) enc_r(op_special, R_0, (rt), (rd), (shamt), fc_sra)
|
||||||
|
|
||||||
|
/* Shift Variable — register-shift forms.
|
||||||
|
* shift_lleft_var(rd, rt, rs) → sllv rd, rt, rs (shamt in low 5 bits of rs)
|
||||||
|
* shift_aright_var(rd, rt, rs) → srav rd, rt, rs */
|
||||||
|
#define shift_lleft_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_sllv)
|
||||||
|
#define shift_aright_var(rd, rt, rs) enc_r(op_special, (rs), (rt), (rd), 0, fc_srav)
|
||||||
|
|
||||||
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
#define shift_lleft_self(rd_rt, shamt) enc_r(op_special, R_0, (rd_rt), (rd_rt), (shamt), fc_sll)
|
||||||
|
|
||||||
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
#define mask_upper(rd, rt, shamt) shift_lleft(rd, rt, shamt), shift_lright(rd, rt, shamt)
|
||||||
@@ -366,20 +375,21 @@ enum { _BitOffsets = 0
|
|||||||
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
* WARNING: `jump(off)` CANNOT BE USED for within-atom jumps in the current pipeline.
|
||||||
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
* The MIPS j opcode encodes `(target_addr >> 2)` in its 26-bit immediate field; an ABSOLUTE byte address, not a relative word offset.
|
||||||
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
* The metaprogram computes `off` as a relative word offset (`target_word_idx - branch_word_idx - 1`), which the assembler/linker does NOT resolve.
|
||||||
*
|
|
||||||
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
* `jump(off)` is only safe when the BUILD PIPELINE owns the absolute position of the emitted code — i.e. when: s
|
||||||
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
* - the build emits a symbol-relative `.word` expression that the linker resolvess via `R_MIPS_26`, OR
|
||||||
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
* - the code is hand-assembled with explicit absolute targets, OR a custom post-build patcher resolves the 26-bit field.
|
||||||
|
* TODO(Ed): Review this.. technically we can resolve aboslute jumps on baked atoms? (Even proedurally generated ones...)
|
||||||
*/
|
*/
|
||||||
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
#define jump(off) enc_i(op_j, R_0, R_0, (off))
|
||||||
|
|
||||||
|
// Annotate an instruction as filling a branch-delay slot.
|
||||||
|
#define BdSlot_
|
||||||
|
|
||||||
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
/* jump_rel off — unconditional relative jump (the within-atom-safe `jump`).
|
||||||
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`.
|
* MIPS I R3000A has no "branch always" opcode. The idiom for an unconditional relative jump is `beq $0, $0, off`. */
|
||||||
*/
|
|
||||||
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
#define jump_rel(off) branch_equal(R_0, R_0, (off))
|
||||||
|
|
||||||
/* call_addr off — jump-and-link to immediate address.
|
/* call_addr off — jump-and-link to immediate address.
|
||||||
*
|
|
||||||
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
* Same WARNING as `jump(off)` above: the jal opcode also encodes an absolute 26-bit target.
|
||||||
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
* For within-atom calls, the current pipeline has no equivalent always-taken call-and-link idiom.
|
||||||
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
* Workaround: `branch_link` (always-taken branch + explicit `la $ra, next_word_addr; jr $ra`), or just use `call_reg($tmp)` after loading the target into a register.
|
||||||
@@ -397,13 +407,7 @@ enum { _BitOffsets = 0
|
|||||||
* sub_s / sub_u → sub / subu
|
* sub_s / sub_u → sub / subu
|
||||||
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
* mult_s / mult_u → mult / multu (writes HI/LO; result in LO)
|
||||||
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
* div_s / div_u → div / divu (LO = quot, HI = rem)
|
||||||
*
|
*/
|
||||||
* NOTE: dsl.h defines `add_s`/`sub_s`/`mut_s`/`gt_s`/etc. as _Generic-based signed integer-arithmetic helpers for U1/U2/U4.
|
|
||||||
* Those live in a different conceptual layer (generic arithmetic on DSL types) and would collide with the instruction encoders here.
|
|
||||||
* The `#undef` below lets the gas-style names below win; if a file needs both, the dsl.h versions can be reached via their long forms
|
|
||||||
* (e.g. `def_signed_op`-style or the underlying `add_s1/s2/s4`). */
|
|
||||||
#undef add_s
|
|
||||||
#undef sub_s
|
|
||||||
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
#define add_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_add)
|
||||||
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
#define add_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_addu)
|
||||||
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
#define sub_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_sub)
|
||||||
@@ -413,6 +417,7 @@ enum { _BitOffsets = 0
|
|||||||
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
#define div_s(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_div)
|
||||||
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
#define div_u(rd, rs, rt) enc_r(op_special, (rs), (rt), (rd), 0, fc_divu)
|
||||||
|
|
||||||
|
// TODO(Ed): Change convention of 'self' to ds for (destination is source)?
|
||||||
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
#define add_u_self(rd_rs, rt) add_u(rd_rs, rd_rs, rt)
|
||||||
|
|
||||||
/* --- Arithmetic I-type (immediate) --- */
|
/* --- Arithmetic I-type (immediate) --- */
|
||||||
@@ -455,9 +460,13 @@ enum { _BitOffsets = 0
|
|||||||
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
#define shift_amount(rd, rt, n) shift_lleft(rd, rt, n)
|
||||||
|
|
||||||
/* nop — sll $0, $0, 0 */
|
/* nop — sll $0, $0, 0 */
|
||||||
#define nop shift_lleft(rdiscard, rdiscard, 0)
|
#define nop shift_lleft(R_0, R_0, 0)
|
||||||
#define nop2 nop, nop
|
#define nop2 nop, nop
|
||||||
|
|
||||||
|
// li_s — load signed 16-bit immediate into GPR (addiu rt, $0, imm — sign-extends).
|
||||||
|
#define li_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
|
// #define load_imm_s(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
|
|
||||||
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
#define load_imm_1w(rt, imm) add_ui((rt), R_0, (imm))
|
||||||
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
#define load_imm_1w_s0(rt, imm) add_si((rt)), R_0, (imm))
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,197 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "mips.h"
|
||||||
|
# include "dsl.atom.h"
|
||||||
|
# include "lottes_tape.h"
|
||||||
|
# include "pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(pad_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pad_set_centered_axes(AtomBuilder_R ab, Reg state, Reg scratch) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
load_upper_i(scratch, (PadAxis_Centered >> 16) & 0xFFFF),
|
||||||
|
or_i_self( scratch, PadAxis_Centered & 0xFFFF),
|
||||||
|
// mac_load_word_imm(scratch, PadAxis_Centered),
|
||||||
|
store_word( scratch, state, O_(PadState,axes)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pad_set_id_byte(AtomBuilder_R ab, Reg state, Reg r_id, U1 id_value) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
add_ui( r_id, R_0, id_value),
|
||||||
|
store_byte(r_id, state, O_(PadState,id)),
|
||||||
|
})
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_pad_set_status(AtomBuilder_R ab, U4 r_tmp, U1 r_state, U4 pad_status) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
add_ui( r_tmp, R_0, pad_status),
|
||||||
|
store_word(r_tmp, r_state, O_(PadState,status)),
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Invert r_buttons (active-low → active-high) and store to PadState.buttons.
|
||||||
|
* r_buttons must already be loaded (the caller is responsible for filling the load-delay slot of
|
||||||
|
* the preceding load_half_u with an instruction that doesn't read r_buttons). */
|
||||||
|
FI_ Slice_MipsCode ac_pad_store_inverted_buttons(AtomBuilder_R ab, U1 r_buttons, U1 r_pad_state) atom_dbg_skip MipsAtomComp_Proc_(ab, {
|
||||||
|
nor_u( r_buttons, r_buttons, R_0),
|
||||||
|
store_half( r_buttons, r_pad_state, O_(PadState,buttons)),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs (Mips Atom Components)
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
/* ----- pad_bios_snapshot -----
|
||||||
|
* Per-frame snapshot of one BIOS pad buffer into PadState.
|
||||||
|
* Decoder (branch ladder on raw[0] status + raw[1] id):
|
||||||
|
* 1. raw[0] == 0xFF -> Disconnected (buttons=0, axes=0x80)
|
||||||
|
* 2. raw[0]==0 && raw[1]==0 -> Pending (buttons=0, axes=0x80)
|
||||||
|
* 3. raw[1] == 0x41 -> Digital (buttons normalized; axes=0x80)
|
||||||
|
* 4. raw[1] == 0x53 -> AnalogStick (buttons normalized; axes from raw[4..7])
|
||||||
|
* 5. raw[1] in 0x7x -> AnalogPad (buttons normalized; axes from raw[4..7])
|
||||||
|
* 6. else -> Unsupported (buttons=0, axes=0x80)
|
||||||
|
*
|
||||||
|
* Buttons normalization: byte_swap16((~raw_buttons) & 0xFFFF).
|
||||||
|
* raw_buttons = load_half_u(raw, 2) = raw[2] | (raw[3] << 8).
|
||||||
|
* byte_swap16(x) = (x >> 8) | (x << 8); nor(x, R_0) = ~x. store_half truncates to 16 bits so the upper-16 mask is implicit in the store.
|
||||||
|
*
|
||||||
|
* Register use (atom-local; no wave-context touched):
|
||||||
|
* R_T0 = raw base : Kept throughout; axes loads read raw[4..7] from R_T0.
|
||||||
|
* R_T1 = state base : Kept throughout; all stores go through R_T1.
|
||||||
|
* R_T2 = raw[0] status : Alive across the disc/pending/id dispatch, then dead.
|
||||||
|
* R_T3 = raw[1] id : Alive across the id dispatch, then dead.
|
||||||
|
* R_T4 = scratch : Shifts, compares, immediate loads, store values.
|
||||||
|
* R_T5 = scratch : Parallel lui + ori for the 0x80808080 axes constant + byte-swap target.
|
||||||
|
*/
|
||||||
|
enum {
|
||||||
|
R_PadRaw = R_T0 atom_reg atom_type(U1),
|
||||||
|
R_PadState = R_T1 atom_reg atom_type(PadState*),
|
||||||
|
R_RawStatus = R_T2 atom_reg,
|
||||||
|
R_RawId = R_T3 atom_reg,
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_PadBiosSnapshot) {
|
||||||
|
PadBiosRaw* raw;
|
||||||
|
PadState* state;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_bios_snapshot) atom_info(atom_bind(Binds_PadBiosSnapshot)
|
||||||
|
, atom_reads( R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||||
|
, atom_writes(R_PadRaw, R_PadState, R_RawStatus, R_RawId)
|
||||||
|
) {
|
||||||
|
/* === Bind consumption: T0 = raw, T1 = state, advance R_TapePtr by 8. */
|
||||||
|
load_word(R_PadRaw, R_TapePtr, O_(Binds_PadBiosSnapshot,raw)),
|
||||||
|
load_word(R_PadState, R_TapePtr, O_(Binds_PadBiosSnapshot,state)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_PadBiosSnapshot)),
|
||||||
|
|
||||||
|
/* === Read raw[0] (status) + raw[1] (id) */
|
||||||
|
load_byte_u(R_RawStatus, R_PadRaw, O_(PadBiosRaw,status)),
|
||||||
|
load_byte_u(R_RawId, R_PadRaw, O_(PadBiosRaw,id)),
|
||||||
|
|
||||||
|
atom_label(snap_root) /* === Case 1: Disconnected (status == 0xFF). */
|
||||||
|
add_ui(R_T4, R_0, PadRawStatus_Timeout), branch_ne(R_RawStatus, R_T4, atom_offset(snap_root, skip_disconnected)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Disconnected. Branch reads R_T4=0xFF in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to pending/id_dispatch), R_T4 is overwritten by the next case body's add_ui — harmless. */
|
||||||
|
|
||||||
|
atom_label(disconnected) /* === Disconnected body. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_Disconnected),
|
||||||
|
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_RawId, PadRawStatus_Timeout),
|
||||||
|
jump_rel(atom_offset(disconnected, snap_end)),
|
||||||
|
/* BD-slot: load next atom's entry point (replaces the nop).
|
||||||
|
* Always jumps to snap_end, where mac_yield_tail() transfers control to R_AtomJmp without re-loading it. */
|
||||||
|
mac_yield_load(),
|
||||||
|
atom_label(skip_disconnected)
|
||||||
|
|
||||||
|
/* === Case 2: Pending (status == 0 && id == 0)
|
||||||
|
* Combined check: if (status | id) != 0 then skip to id_dispatch. Falls through to the Pending case only when both are zero. */
|
||||||
|
or_u_self(R_RawStatus, R_RawId), branch_ne(R_RawStatus, R_0, atom_offset(case_2, id_dispatch)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Pending. Branch reads R_RawStatus in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to id_dispatch), R_T4 is overwritten by the digital/analog body add_ui - harmless. */
|
||||||
|
|
||||||
|
atom_label(pending) /* === Pending body (status=0, id=0 — pre-IRQ-empty buffer). */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_Pending),
|
||||||
|
store_half( R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||||
|
store_byte(R_RawId, R_PadState, O_(PadState,id)),
|
||||||
|
jump_rel(atom_offset(pending, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(id_dispatch) /* === Case 3-6: ID dispatch */
|
||||||
|
add_ui(R_T4, R_0, PadRawId_Digital), branch_ne(R_RawId, R_T4, atom_offset(id_dispatch, try_analog_stick)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_Digital. Branch reads R_RawId in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_analog_stick), R_T4 is overwritten by the analog body add_ui. */
|
||||||
|
|
||||||
|
/* === Digital body (status, buttons normalize, axes=0x80, id, branch.
|
||||||
|
* R_T5 holds the 0x80808080 axes constant (loaded into the load-delay slot of the buttons-load).
|
||||||
|
* R_T5 is then "dead" — only consumed at the analog_pad range check downstream. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_Digital),
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw, buttons)), /* R_T4 = raw_buttons; */
|
||||||
|
mac_load_word_imm(R_T5, PadAxis_Centered), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||||
|
// load_upper_i(R_T5, PadAxis_Centered_Hi), or_i_self(R_T5, PadAxis_Centered_Lo),
|
||||||
|
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||||
|
store_word(R_T5, R_PadState, O_(PadState, axes)), /* single sw writes the 4-byte axes block at offset 8 (left_x, left_y, right_x, right_y) */
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_T4, PadRawId_Digital),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(id_dispatch, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_analog_stick) /* === Case 4: AnalogStick (id == 0x53)*/
|
||||||
|
add_ui(R_T4, R_0, PadRawId_AnalogStick), branch_ne(R_RawId, R_T4, atom_offset(try_analog_stick, try_analog_pad)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_AnalogStick. Branch reads R_RawId in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_analog_pad), R_T4 is overwritten by the analog_pad body add_ui. */
|
||||||
|
|
||||||
|
atom_label(analog_stick) /* === AnalogStick body
|
||||||
|
* R_T5 holds left_xy (loaded into the load-delay slot of the buttons-load via the left-axis load_half_u).
|
||||||
|
* R_T4 holds right_xy (loaded into the load-delay slot of the left-load).
|
||||||
|
* R_T5 is then "dead" — reused for the id-byte value load in mac_pad_write_id_byte.
|
||||||
|
* The buttons invert+store happens BEFORE R_T4 is overwritten by the right_xy load. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogStick),
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||||
|
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||||
|
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 (was buttons) with right_xy */
|
||||||
|
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_T5, PadRawId_AnalogStick),
|
||||||
|
jump_rel(atom_offset(analog_stick, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_analog_pad) /* === Case 5-6: AnalogPad (id & 0xF0 == 0x70) */
|
||||||
|
and_i( R_T4, R_RawId, PadRawId_AnalogPadMask),
|
||||||
|
add_ui( R_T5, R_0, PadRawId_AnalogPadValue),
|
||||||
|
branch_ne(R_T4, R_T5, atom_offset(try_analog_pad, try_unsupported)),
|
||||||
|
/* BD-slot: pre-compute PadStatus_AnalogPad. Branch reads R_T4 in EX before this WB completes.
|
||||||
|
* If branch NOT taken (fall through to try_unsupported), R_T4 is overwritten by the unsupported body add_ui. */
|
||||||
|
|
||||||
|
atom_label(analog_pad) /* === AnalogPad body
|
||||||
|
* Same shape as AnalogStick with AnalogPad status. R_T5 holds left_xy (it's dead on this path).
|
||||||
|
* The id byte is raw id from the BIOS buffer (R_RawId already holds raw[1]).
|
||||||
|
* Buttons invert + store happens before R_T4 is overwritten by the right_xy load. */
|
||||||
|
mac_pad_set_status(R_T4, R_PadState, PadStatus_AnalogPad),
|
||||||
|
load_half_u( R_T4, R_PadRaw, O_(PadBiosRaw,buttons)), /* R_T4 = raw_buttons; delay slot at the next instruction */
|
||||||
|
load_half_u( R_T5, R_PadRaw, O_(PadBiosRaw,left)), /* fills the buttons-load's delay slot (doesn't read R_T4) */
|
||||||
|
mac_pad_store_inverted_buttons(R_T4, R_PadState), /* R_T4 settled: nor + sh writes ~raw_buttons to state.buttons */
|
||||||
|
load_half_u(R_T4, R_PadRaw, O_(PadBiosRaw,right)), /* fills R_T5's load-delay slot (doesn't read R_T5); overwrites R_T4 with right_xy */
|
||||||
|
store_half( R_T5, R_PadState, O_(PadState, left)),
|
||||||
|
store_half( R_T4, R_PadState, O_(PadState, right)),
|
||||||
|
store_byte( R_RawId, R_PadState, O_(PadState, id)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(analog_pad, snap_end)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the AnalogPad range-check miss. */
|
||||||
|
add_ui( R_T4, R_0, PadStatus_Unsupported),
|
||||||
|
store_word(R_T4, R_PadState, O_(PadState,status)),
|
||||||
|
store_half(R_0, R_PadState, O_(PadState,buttons)),
|
||||||
|
mac_pad_set_centered_axes(R_PadState, R_T4),
|
||||||
|
mac_pad_set_id_byte(R_PadState, R_RawId, PadUnknownId_Sentinel),
|
||||||
|
/* Fall through to snap_end. */
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(snap_end)
|
||||||
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "dsl.h"
|
||||||
|
# include "gcc_asm.h"
|
||||||
|
# include "mips.h"
|
||||||
|
# include "bios.h"
|
||||||
|
# include "pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
/* Uses ONE 8-byte frame allocated via the compiler's standard prologue.
|
||||||
|
* 4 wasted-arg words for B(12h) InitPAD2 are at [SP+0..15] but are not explicitly allocated.
|
||||||
|
* Compiler handles the MIPS O32 "wasted stack" convention for us by treating the B-call as a 4-arg call.
|
||||||
|
*
|
||||||
|
* The buffer pointers are passed as arguments so the compiler keeps them in callee-saved registers;
|
||||||
|
* The B(12h) asm volatile block does NOT clobber those registers (it clobbers only the volatile GPRs + B-table arg registers explicitly).
|
||||||
|
* The C-level writes after the call re-load the pointers from their callee-saved homes.
|
||||||
|
*
|
||||||
|
* The clobber list for both B-calls names the full BIOS destroy set documented in kernelbios.md:167-174 (R1..R15, R24..R25, R31, HI/LO).
|
||||||
|
* The kernel-ABI "volatile GPRs" subset is clb_mem_drain; the rest of the destroy set is enumerated explicitly here. */
|
||||||
|
NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
||||||
|
{
|
||||||
|
/* Pin raw0 + raw1 to $a0 + $a1 via rgcc; the B(12h) call uses these directly.
|
||||||
|
* The `(void)` casts mark them as unread after the call so the compiler doesn't need to move them back. */
|
||||||
|
register PadBiosRaw* p0 rgcc(R_A0) = raw0;
|
||||||
|
register PadBiosRaw* p1 rgcc(R_A1) = raw1;
|
||||||
|
(void)p0; (void)p1;
|
||||||
|
|
||||||
|
// TODO(Ed): Properly annotate the raw values in the inline asm instructions.
|
||||||
|
// Use enums.
|
||||||
|
|
||||||
|
/* B(12h) InitPAD2(raw0, 0x22, raw1, 0x22)
|
||||||
|
* $a0 = raw0 (rgcc-bound; survives the sequence below)
|
||||||
|
* $a1 = raw1 (preserved into $a2 before $a1 is overwritten)
|
||||||
|
* $a2 = raw1 (moved from $a1; survives $a1's overwrite)
|
||||||
|
* $a3 = 0x22 (immediate)
|
||||||
|
* $t1 = 0x12 (function number)
|
||||||
|
* $t2 = 0xB0 (BIOS B-table address) */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
or_u( R_A2, R_A1, R_0), /* $a2 = $a1 = raw1 */
|
||||||
|
add_ui( R_A1, R_0, bios_pad_buffer_size), /* $a1 = 0x22 */
|
||||||
|
add_ui( R_A3, R_0, bios_pad_buffer_size), /* $a3 = 0x22 */
|
||||||
|
add_ui( R_T1, R_0, bios_init_pad_2), /* $t1 = 0x12 */
|
||||||
|
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 */
|
||||||
|
call_reg(R_T2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_rpins, r_use(p0), r_use(p1)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
|
||||||
|
/* The C-level writes re-load the pointers via the parameter names and write 0xFF to each
|
||||||
|
* buffer's status byte to mark the initial-state hazard documented in kernelbios.md:1621-1624. */
|
||||||
|
u1_v(raw0)[0] = 0xFF;
|
||||||
|
u1_v(raw1)[0] = 0xFF;
|
||||||
|
|
||||||
|
/* B(13h) StartPAD2() — no args. The BIOS preserves $sp. */
|
||||||
|
asm volatile(
|
||||||
|
asm_words(
|
||||||
|
add_ui( R_T1, R_0, bios_start_pad_2), /* $t1 = 0x13 */
|
||||||
|
add_ui( R_T2, R_0, bios_btable_addr), /* $t2 = 0xB0 (re-load) */
|
||||||
|
call_reg(R_T2), /* jalr $t2, $ra */
|
||||||
|
nop /* BD slot */
|
||||||
|
)
|
||||||
|
asm_clobber:
|
||||||
|
rlit(R_AT),
|
||||||
|
rlit(R_V0), rlit(R_V1),
|
||||||
|
rlit(R_T0), rlit(R_T1), rlit(R_T2), rlit(R_T3), rlit(R_T4),
|
||||||
|
rlit(R_T5), rlit(R_T6), rlit(R_T7), rlit(R_T8), rlit(R_T9),
|
||||||
|
rlit(R_RA),
|
||||||
|
clb_mem_drain
|
||||||
|
);
|
||||||
|
}
|
||||||
+79
-37
@@ -1,28 +1,30 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# pragma once
|
# pragma once
|
||||||
# include "dsl.h"
|
# include "dsl.h"
|
||||||
|
# include "math.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
/* PSX button bit positions — 1:1 with PSX-SPX docs at docs/psx-spx/docs/controllersandmemorycards.md:405-421.
|
||||||
* Wire is active-low (0 = pressed).
|
* Wire is active-low (0 = pressed).
|
||||||
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF; the active-low-to-active-high inversion is applied bit-by-bit. */
|
* The decoder atom computes buttons = (~raw_buttons) & 0xFFFF;
|
||||||
enum {
|
* active-low-to-active-high inversion is applied bit-by-bit. */
|
||||||
Bit_(Pad_Select, 0),
|
typedef Enum_(U2, PadBtns) {
|
||||||
Bit_(Pad_L3, 1),
|
Bit_(Pad_Select, 0),
|
||||||
Bit_(Pad_R3, 2),
|
Bit_(Pad_L3, 1),
|
||||||
Bit_(Pad_Start, 3),
|
Bit_(Pad_R3, 2),
|
||||||
Bit_(Pad_Up, 4),
|
Bit_(Pad_Start, 3),
|
||||||
Bit_(Pad_Right, 5),
|
Bit_(Pad_Up, 4),
|
||||||
Bit_(Pad_Down, 6),
|
Bit_(Pad_Right, 5),
|
||||||
Bit_(Pad_Left, 7),
|
Bit_(Pad_Down, 6),
|
||||||
Bit_(Pad_L2, 8),
|
Bit_(Pad_Left, 7),
|
||||||
Bit_(Pad_R2, 9),
|
Bit_(Pad_L2, 8),
|
||||||
Bit_(Pad_L1, 10),
|
Bit_(Pad_R2, 9),
|
||||||
Bit_(Pad_R1, 11),
|
Bit_(Pad_L1, 10),
|
||||||
|
Bit_(Pad_R1, 11),
|
||||||
Bit_(Pad_Triangle, 12),
|
Bit_(Pad_Triangle, 12),
|
||||||
Bit_(Pad_Circle, 13),
|
Bit_(Pad_Circle, 13),
|
||||||
Bit_(Pad_Cross, 14),
|
Bit_(Pad_Cross, 14),
|
||||||
Bit_(Pad_Square, 15),
|
Bit_(Pad_Square, 15),
|
||||||
};
|
};
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
@@ -32,18 +34,22 @@ enum {
|
|||||||
Pad1 = 1 << PadId_Offset,
|
Pad1 = 1 << PadId_Offset,
|
||||||
};
|
};
|
||||||
|
|
||||||
#define pad0_(btn_id) (btn_id << Pad0)
|
/* =============================================================================
|
||||||
#define pad1_(btn_id) (btn_id << Pad1)
|
|
||||||
|
|
||||||
/* ============================================================
|
|
||||||
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
* BIOS pad-buffer subsystem: docs/psx-spx/docs/kernelbios.md (B(12h) + B(13h))
|
||||||
* ============================================================ */
|
* ============================================================================= */
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
PAD_BIOS_RAW_SIZE = 0x22,
|
PAD_BIOS_RAW_SIZE = 0x22,
|
||||||
};
|
};
|
||||||
|
// BIOS pad buffer layout (docs/psx-spx/docs/kernelbios.md (InitPAD2 returns 0x22 = 34 bytes per port)).
|
||||||
|
// Bytes 0..7 are the named snapshot region; bytes 8..33 are reserved (the BIOS writes the buffer raw; we only read bytes 0..7 via O_(PadBiosRaw, ...)).
|
||||||
typedef Struct_(PadBiosRaw) {
|
typedef Struct_(PadBiosRaw) {
|
||||||
U1 bytes[PAD_BIOS_RAW_SIZE];
|
U1 status; /* offset 0 (PadRawStatus_Ok / PadRawStatus_Timeout) */
|
||||||
|
U1 id; /* offset 1 (PadRawId_Digital / PadRawId_AnalogStick / 0x7x AnalogPad) */
|
||||||
|
U2 buttons; /* offset 2-3 (active-low 16-bit button map) */
|
||||||
|
V2_U1 right; /* offset 4-5 (right stick x, y) */
|
||||||
|
V2_U1 left; /* offset 6-7 (left stick x, y) */
|
||||||
|
U1 reserved[PAD_BIOS_RAW_SIZE - 8]; /* offset 8..33 */
|
||||||
};
|
};
|
||||||
|
|
||||||
typedef Enum_(U4, PadStatus) {
|
typedef Enum_(U4, PadStatus) {
|
||||||
@@ -56,18 +62,54 @@ typedef Enum_(U4, PadStatus) {
|
|||||||
PadStatus_Invalid,
|
PadStatus_Invalid,
|
||||||
};
|
};
|
||||||
|
|
||||||
/* PadState — per-port normalized runtime state.
|
/* Distinct from the game-facing PadStatus enum: PadRawStatus_Ok and PadRawStatus_Timeout are raw BIOS values;
|
||||||
* Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
* PadStatus_* are game-facing post-decode states. PadUnknownId_Sentinel is written by the decoder
|
||||||
* form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
* when the controller id does not match any known controller type.
|
||||||
* The struct size stays 12 bytes (unchanged from the prior order,
|
* PadAxisCentered_Word: Four-byte 0x80 pattern used to clear / center
|
||||||
* which left the C compiler to insert 1 byte of trailing pad to reach the 4-byte struct alignment). */
|
* four byte axes at PadState.left_x through PadState.right_y. */
|
||||||
typedef Struct_(PadState) {
|
typedef Enum_(U1, PadRawStatus) {
|
||||||
PadStatus status; /* offset 0, size 4 (U4) */
|
PadRawStatus_Ok = 0x00,
|
||||||
U2 buttons; /* offset 4, size 2 */
|
PadRawStatus_Timeout = 0xFF,
|
||||||
U1 id; /* offset 6, size 1 */
|
|
||||||
U1 pad; /* offset 7, size 1 — explicit pad to align the axes block */
|
|
||||||
U1 left_x; /* offset 8, size 1 — store_word target (4-byte aligned) */
|
|
||||||
U1 left_y; /* offset 9, size 1 */
|
|
||||||
U1 right_x; /* offset 10, size 1 */
|
|
||||||
U1 right_y; /* offset 11, size 1 */
|
|
||||||
};
|
};
|
||||||
|
typedef Enum_(U1, PadRawId) {
|
||||||
|
PadRawId_Digital = 0x41,
|
||||||
|
PadRawId_AnalogStick = 0x53,
|
||||||
|
PadRawId_AnalogPadMask = 0xF0,
|
||||||
|
PadRawId_AnalogPadValue = 0x70,
|
||||||
|
};
|
||||||
|
typedef Enum_(U1, PadUnknownId) {
|
||||||
|
PadUnknownId_Sentinel = 0xFF,
|
||||||
|
};
|
||||||
|
typedef Enum_(U4, PadAxisCentered) {
|
||||||
|
PadAxis_Centered_Hi = 0x8080,
|
||||||
|
PadAxis_Centered_Lo = 0x8080,
|
||||||
|
PadAxis_Centered = 0x80808080U,
|
||||||
|
};
|
||||||
|
typedef Enum_(U1, PadDeadZone) {
|
||||||
|
PadDeadZone_LowBound = 0x70, /* left_x < LowBound → active; delta = 0x80 - left_x > 0 (rightward pull) */
|
||||||
|
PadDeadZone_Center = 0x80, /* analog rest position; left_x == Center → delta = 0 (no rotation) */
|
||||||
|
PadDeadZone_HighBound = 0x90, /* left_x > HighBound → active; delta = 0x80 - left_x < 0 (leftward pull) */
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
typedef Struct_(PadAxes) {
|
||||||
|
V2_U1 left; /* offset 8-9 */
|
||||||
|
V2_U1 right; /* offset 10-11 */
|
||||||
|
};
|
||||||
|
// Field order is chosen so that the 4 axes (left_x, left_y, right_x, right_y)
|
||||||
|
// form a contiguous 4-byte block at offset 8, allowing a single `store_word` to clear-or-write all 4 axes in one MIPS instruction.
|
||||||
|
typedef Struct_(PadState) {
|
||||||
|
PadStatus status; /* offset 0, (U4) */
|
||||||
|
PadBtns buttons; /* offset 4, */
|
||||||
|
U1 id; /* offset 6, */
|
||||||
|
byte_pad(1); /* offset 7, explicit pad to align the axes block */
|
||||||
|
union {
|
||||||
|
A2_V2_U1 axes; /* offset 8-11 store_target (4-byte aligned)*/
|
||||||
|
struct {
|
||||||
|
V2_U1 left; /* offset 8-9 */
|
||||||
|
V2_U1 right; /* offset 10-11 */
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
internal void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1);
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "psyq.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(pysq_atom_c);
|
||||||
@@ -1,8 +1,8 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# pragma once
|
# pragma once
|
||||||
# include "duffle/dsl.h"
|
# include "dsl.h"
|
||||||
# include "duffle/math.h"
|
# include "math.h"
|
||||||
# include "duffle/gp.h"
|
# include "gp.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
typedef Struct_(DrawEnv_Packed) { U4 tag; U4 code[15]; };
|
||||||
@@ -64,9 +64,9 @@ typedef Struct_(Tile) {
|
|||||||
Linear Algebra
|
Linear Algebra
|
||||||
*/
|
*/
|
||||||
|
|
||||||
M3_S2* m3s2_rotation (V3_S2* vec, M3_S2* mat) asm("RotMatrix");
|
MT3_S2S4* mt3s2s4_rotation (V3_S2* vec, MT3_S2S4* mat) asm("RotMatrix");
|
||||||
M3_S2* m3s2_translation(M3_S2* mat, V3_S4* vec) asm("TransMatrix");
|
MT3_S2S4* mt3s2s4_translation(MT3_S2S4* mat, V3_S4* vec) asm("TransMatrix");
|
||||||
M3_S2* m3s2_scale (M3_S2* mat, V3_S4* vec) asm("ScaleMatrix");
|
MT3_S2S4* mt3s2s4_scale (MT3_S2S4* mat, V3_S4* vec) asm("ScaleMatrix");
|
||||||
|
|
||||||
// Rotation, Translation, Perspective
|
// Rotation, Translation, Perspective
|
||||||
|
|
||||||
@@ -99,5 +99,23 @@ FI_ S4 rtp_avg_nclip_a4_v3s2(
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
void gte_matrix_set_rotation (M3_S2* mat) asm("SetRotMatrix");
|
void gte_matrix_set_rotation (MT3_S2S4* mat) asm("SetRotMatrix");
|
||||||
void gte_matrix_set_translation(M3_S2* mat) asm("SetTransMatrix");
|
void gte_matrix_set_translation(MT3_S2S4* mat) asm("SetTransMatrix");
|
||||||
|
|
||||||
|
// Einheit, Metrication to unit vector. "Normalization", not Orthogonal "Normal, Normalis". Directionalization.
|
||||||
|
// RGA(Lengyel): Normalize the bulk of a zero-weight direction. This is not finite-point unitization (which forces w=1).
|
||||||
|
S4 normalize_v3s4(V3_S4* v0, V3_S4* v1) asm("VectorNormal");
|
||||||
|
|
||||||
|
// RGA(Lengyel): Apply the matrix expansion of a rigid transformation.
|
||||||
|
// Motor antiproduct is equivalent for unitized points; LA form is what GTE consumes.
|
||||||
|
V3_S4* mul_m3s2_v3s4(MT3_S2S4* m, V3_S4* v, V3_S4* result) asm("ApplyMatrixLV");
|
||||||
|
|
||||||
|
// RGA(Lengyel): Store the full translation column. The motor translator would store half this displacement in m.xyz.
|
||||||
|
MT3_S2S4* trans_m3s2(MT3_S2S4* m, V3_S4* off) asm("TransMatrix");
|
||||||
|
|
||||||
|
MT3_S2S4* gte_comp_coord_m3s2(MT3_S2S4* m0, MT3_S2S4* m1, MT3_S2S4* result) asm("CompMatrixLV");
|
||||||
|
|
||||||
|
// RGA(Lengyel): Complement(Wedge(a,b)), i.e. the Euclidean 3D complement of the exterior product, stored as a V3_S4.
|
||||||
|
// The underlying GTE OP is a specialized signed-16-bit D x IR command; the wedge interpretation is a 3D dual of the same 3 scalars.
|
||||||
|
void cross_v3s4(V3_S4* v0, V3_S4* v1, V3_S4* result) asm("OuterProduct12");
|
||||||
|
|
||||||
@@ -6,7 +6,7 @@
|
|||||||
// One line per macro that appears in your atom sources.
|
// One line per macro that appears in your atom sources.
|
||||||
//
|
//
|
||||||
// This file is encoding-macros-only.
|
// This file is encoding-macros-only.
|
||||||
// The auto-generated component macros (mac_X) live in duffle/gen/<dir>.macs.h (included separately by the unity build).
|
// The auto-generated component macros (mac_X) live in the source directory's own gen/macs.h (per-directory aggregation; included separately by the unity build).
|
||||||
// The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped
|
// The unity build should include THIS file and the .macs.h file in the same TU, with both wrapped
|
||||||
// (or the include guard order handled) to avoid WORD_COUNT redeclaration.
|
// (or the include guard order handled) to avoid WORD_COUNT redeclaration.
|
||||||
//
|
//
|
||||||
@@ -15,6 +15,8 @@
|
|||||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
|
||||||
WORD_COUNT(nop, 1)
|
WORD_COUNT(nop, 1)
|
||||||
|
WORD_COUNT(atom_label, 0)
|
||||||
|
WORD_COUNT(atom_offset, 0)
|
||||||
WORD_COUNT(load_upper_i, 1)
|
WORD_COUNT(load_upper_i, 1)
|
||||||
WORD_COUNT(jump_reg, 1)
|
WORD_COUNT(jump_reg, 1)
|
||||||
WORD_COUNT(jump_link, 1)
|
WORD_COUNT(jump_link, 1)
|
||||||
@@ -54,6 +56,15 @@ WORD_COUNT(gte_sw, 1)
|
|||||||
WORD_COUNT(gte_cmdw_rtpt, 1)
|
WORD_COUNT(gte_cmdw_rtpt, 1)
|
||||||
WORD_COUNT(gte_cmdw_nclip, 1)
|
WORD_COUNT(gte_cmdw_nclip, 1)
|
||||||
WORD_COUNT(gte_avg_sort_z3, 1)
|
WORD_COUNT(gte_avg_sort_z3, 1)
|
||||||
|
WORD_COUNT(gte_cmdw_sqr, 1)
|
||||||
|
WORD_COUNT(gte_cmdw_gpf, 1)
|
||||||
|
WORD_COUNT(shift_lleft_var, 1)
|
||||||
|
WORD_COUNT(shift_aright_var, 1)
|
||||||
|
WORD_COUNT(li_s, 1)
|
||||||
|
WORD_COUNT(and_i, 1)
|
||||||
|
WORD_COUNT(add_si, 1)
|
||||||
|
WORD_COUNT(branch_lt_zero, 1)
|
||||||
|
WORD_COUNT(sub_s, 1)
|
||||||
WORD_COUNT(sub_u, 1)
|
WORD_COUNT(sub_u, 1)
|
||||||
WORD_COUNT(nop2, 2)
|
WORD_COUNT(nop2, 2)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_camera
|
||||||
|
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.c
|
||||||
|
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.h
|
||||||
|
// source: C:/projects/Pikuma/ps1/code/hello_camera/hello_camera.atom.c
|
||||||
|
// Per-phase register allocations resolved by the lua pass.
|
||||||
|
// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory.
|
||||||
|
|
||||||
|
#define R_GpTmp_Code R_V0_Code
|
||||||
|
|
||||||
@@ -2,53 +2,23 @@
|
|||||||
#pragma once
|
#pragma once
|
||||||
#endif
|
#endif
|
||||||
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||||
// Source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.tape.c
|
// Directory: C:\projects\Pikuma\ps1\code\hello_camera/
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
|
||||||
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||||
|
|
||||||
#ifndef WORD_COUNT
|
#ifndef WORD_COUNT
|
||||||
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_load_v2s2(rs_x, rs_y, r_base, offset) \
|
|
||||||
load_half( rs_x, r_base, O_(V3_S2,x)) \
|
|
||||||
, load_half( rs_y, r_base, O_(V3_S2,y))
|
|
||||||
WORD_COUNT(mac_load_v2s2, 2)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_store_v2s2(rt_x, rt_y, base, offset) \
|
|
||||||
store_half(rt_x, base, offset + O_(V2_S2,x)) \
|
|
||||||
, store_half(rt_y, base, offset + O_(V2_S2,y))
|
|
||||||
WORD_COUNT(mac_store_v2s2, 2)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_store_rects2(rt_x, rt_y, rt_width, rt_height, base, offset) \
|
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)) \
|
|
||||||
, store_half(rt_y, base, offset + O_(Rect_S2,y)) \
|
|
||||||
, store_half(rt_width, base, offset + O_(Rect_S2,width)) \
|
|
||||||
, store_half(rt_height, base, offset + O_(Rect_S2,height))
|
|
||||||
WORD_COUNT(mac_store_rects2, 4)
|
|
||||||
|
|
||||||
/* atom_dbg_skip */
|
|
||||||
#define mac_store_rgb8(rr, rg, rb, base, offset) \
|
|
||||||
store_byte(rr, base, offset + O_(DrawEnv,initial_bg_color.r)) \
|
|
||||||
, store_byte(rg, base, offset + O_(DrawEnv,initial_bg_color.g)) \
|
|
||||||
, store_byte(rb, base, offset + O_(DrawEnv,initial_bg_color.b))
|
|
||||||
WORD_COUNT(mac_store_rgb8, 3)
|
|
||||||
|
|
||||||
#define mac_gcmd_push(cmd, reg_transfer, reg_base, port) \
|
|
||||||
load_upper_i(reg_transfer, cmd >> 16) \
|
|
||||||
, or_i_self( reg_transfer, cmd & 0xFFFF) \
|
|
||||||
, store_word( reg_transfer, reg_base, port)
|
|
||||||
WORD_COUNT(mac_gcmd_push, 3)
|
|
||||||
|
|
||||||
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||||
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||||
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||||
WORD_COUNT(mac_put_disp_env, 15)
|
WORD_COUNT(mac_put_disp_env, 5)
|
||||||
|
|
||||||
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||||
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||||
@@ -67,5 +37,5 @@ WORD_COUNT(mac_put_disp_env, 15)
|
|||||||
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||||
WORD_COUNT(mac_put_draw_env, 48)
|
WORD_COUNT(mac_put_draw_env, 16)
|
||||||
|
|
||||||
@@ -0,0 +1,68 @@
|
|||||||
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_camera\
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_camera\hello_camera.atom.c
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#pragma region hello_camera
|
||||||
|
|
||||||
|
|
||||||
|
// --- atom: pad_input_cube_rotation (60 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||||
|
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||||
|
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||||
|
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||||
|
#define _atom_offset_dead_zone_skip_exit_stick 24
|
||||||
|
#define _atom_offset_end_low_exit_stick 12
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||||
|
atom_offset_dpad_right_exit_dpad_right = _atom_offset_dpad_right_exit_dpad_right,
|
||||||
|
atom_offset_dead_zone_low_check_dead_low_active = _atom_offset_dead_zone_low_check_dead_low_active,
|
||||||
|
atom_offset_dead_zone_high_check_dead_high_active = _atom_offset_dead_zone_high_check_dead_high_active,
|
||||||
|
atom_offset_dead_zone_skip_exit_stick = _atom_offset_dead_zone_skip_exit_stick,
|
||||||
|
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: pad_input_cam (40 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_left_x_exit_left_x 3
|
||||||
|
#define _atom_offset_right_x_exit_right_x 3
|
||||||
|
#define _atom_offset_up_y_exit_up_y 3
|
||||||
|
#define _atom_offset_down_y_exit_down_y 3
|
||||||
|
#define _atom_offset_cross_z_exit_cross_z 3
|
||||||
|
#define _atom_offset_circle_z_exit_circle_z 3
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_left_x_exit_left_x = _atom_offset_left_x_exit_left_x,
|
||||||
|
atom_offset_right_x_exit_right_x = _atom_offset_right_x_exit_right_x,
|
||||||
|
atom_offset_up_y_exit_up_y = _atom_offset_up_y_exit_up_y,
|
||||||
|
atom_offset_down_y_exit_down_y = _atom_offset_down_y_exit_down_y,
|
||||||
|
atom_offset_cross_z_exit_cross_z = _atom_offset_cross_z_exit_cross_z,
|
||||||
|
atom_offset_circle_z_exit_circle_z = _atom_offset_circle_z_exit_circle_z,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: cube_g4_face (76 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_cull_cube_g4_face_exit 41
|
||||||
|
#define _atom_offset_bounds_chk_cube_g4_face_exit 24
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_cull_cube_g4_face_exit = _atom_offset_cull_cube_g4_face_exit,
|
||||||
|
atom_offset_bounds_chk_cube_g4_face_exit = _atom_offset_bounds_chk_cube_g4_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- atom: floor_f3_face (58 words) ---
|
||||||
|
|
||||||
|
#define _atom_offset_culling_floor_f3_face_exit 25
|
||||||
|
#define _atom_offset_bounds_chk_floor_f3_face_exit 16
|
||||||
|
|
||||||
|
enum {
|
||||||
|
atom_offset_culling_floor_f3_face_exit = _atom_offset_culling_floor_f3_face_exit,
|
||||||
|
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion hello_camera
|
||||||
|
|
||||||
@@ -0,0 +1,961 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
# include "duffle/dsl.atom.h"
|
||||||
|
# include "duffle/lottes_tape.h"
|
||||||
|
# include "duffle/mips.h"
|
||||||
|
# include "duffle/gte.h"
|
||||||
|
# include "duffle/gp.h"
|
||||||
|
# include "duffle/pad.h"
|
||||||
|
# include "duffle/word_count.metadata.h"
|
||||||
|
# include "duffle/psyq.h"
|
||||||
|
# include "duffle/math.atom.c"
|
||||||
|
# include "duffle/mips.atom.c"
|
||||||
|
# include "duffle/gte.atom.c"
|
||||||
|
# include "duffle/gp.atom.c"
|
||||||
|
# include "duffle/psyq.atom.c"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/auto_reg.h"
|
||||||
|
# include "hello_camera.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||||
|
|
||||||
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
|
FI_ Slice_MipsCode ac_put_disp_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
I_ Slice_MipsCode ac_put_draw_env(AtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
|
MipsAtomComp_Proc_(ab, {
|
||||||
|
/*
|
||||||
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
|
* References:
|
||||||
|
* - libpsyx source: `toolchain/psyq-4_7/lib/libgpu.a` (binary, function `PutDrawEnv`)
|
||||||
|
* - PSX-SPX doc: https://problemkaputt.de/psx-spx.htm#gputdrawingcommands
|
||||||
|
* - PSYQ SDK: `setdrawenv` / `makelongdr_env` source
|
||||||
|
* - NOCASH PSX spec: §"GP0(E1h) Draw Mode setting" through §"DR_ENV"
|
||||||
|
*
|
||||||
|
* The 16-word format is documented in the PSYQ SDK manual and on NOCASH's PSX-spec.txt. The libpsyx reference is at:
|
||||||
|
* ./toolchain/psyq-4_7/lib/libgpu.a
|
||||||
|
* (binary; the PutDrawEnv implementation builds the 16-word DR_ENV from the user's DRAWENV struct and emits it via GP0 GPU commands.)
|
||||||
|
*
|
||||||
|
* Word indices (libpsyx PutDrawEnv / SetDrawEnv order):
|
||||||
|
* tag = (length << 24) | addr — 16-word packet (1 tag + 15 code)
|
||||||
|
* code[0] = DrawMode (dfe=1, dtd=0, tpage=0) — must come first per libpsyx
|
||||||
|
* code[1] = TextureWindow (tw=(0,0)) — bare-cmd word; GPU uses current state
|
||||||
|
* code[2] = DrawArea top-left (clip.x=0, clip.y=240)
|
||||||
|
* code[3] = DrawArea bottom-right (clip.x+w=320, clip.y+h=480)
|
||||||
|
* code[4] = DrawOffset (ofs=(0,0)) — bare-cmd word
|
||||||
|
* code[5] = Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit
|
||||||
|
* code[6] = Initial-bg-color (isbg=1, r=7, g=7, b=7)
|
||||||
|
* code[7] = DrawMode (isbg=1, tpage=0) — re-asserts DrawMode with isbg
|
||||||
|
* code[8..10] = padding (NOP) — 3 words to fill the packet
|
||||||
|
* code[11..12] = TextureWindow bottom-right — defaults to (0,0,0,0)
|
||||||
|
* code[13..14] = padding (NOP) — completes the 16-word packet
|
||||||
|
*/
|
||||||
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port), /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */
|
||||||
|
mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port), /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port), /* code[1] TextureWindow (tw=(0,0)) */
|
||||||
|
mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port), /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port), /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */
|
||||||
|
|
||||||
|
mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port), /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port), /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port), /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */
|
||||||
|
mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port), /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */
|
||||||
|
|
||||||
|
/* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
|
||||||
|
/* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port),
|
||||||
|
|
||||||
|
/* code[13..14] Padding (NOP) — completes the 16-word packet. */
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port),
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion MACs
|
||||||
|
|
||||||
|
#pragma region Atom Procs
|
||||||
|
// Modular Atoms
|
||||||
|
|
||||||
|
enum {
|
||||||
|
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
|
||||||
|
R_ResolveScratch = R_T4 atom_reg atom_type(U4*),
|
||||||
|
#define R_ResolveScratch_Code R_T4_Code
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_ResolveLookAt) {
|
||||||
|
MT3_S2S4* look_at;
|
||||||
|
P3_S4* eye;
|
||||||
|
P3_S4* target;
|
||||||
|
V3_S4* up_in;
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ─── ResolveLookAtScratch — offset schema for the resolve_look_at bundle's */
|
||||||
|
typedef Struct_(ResolveLookAtScratch) {
|
||||||
|
V3_S4 fwd; /* offset +0 (16 bytes — 4 S4 fields incl. internal pad) */
|
||||||
|
V3_S4 uz; /* offset +16 (16 bytes) */
|
||||||
|
V3_S4 right; /* offset +32 (16 bytes) */
|
||||||
|
V3_S4 ux; /* offset +48 (16 bytes) */
|
||||||
|
V3_S4 up; /* offset +64 (16 bytes) */
|
||||||
|
V3_S4 uy; /* offset +80 (16 bytes) */
|
||||||
|
P3_S4 eye; /* offset +96 (16 bytes; storage alias of V3_S4) */
|
||||||
|
P3_S4 target; /* offset +112 (16 bytes; storage alias of V3_S4) */
|
||||||
|
V3_S4 up_in; /* offset +128 (16 bytes) */
|
||||||
|
};
|
||||||
|
|
||||||
|
/* ─── resolve_look_at bundle chain atoms ──────────────────────────── */
|
||||||
|
|
||||||
|
typedef Struct_(Binds_ResolveLookAtSub) {
|
||||||
|
P3_S4* target; /* U4 (C-side P3_S4* — read by atom 0 directly; NOT a scratchpad address) */
|
||||||
|
P3_S4* eye; /* U4 (C-side P3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||||
|
V3_S4* up_in; /* U4 (C-side V3_S4* — read by atom 0 directly; staged into scratchpad by atom 0) */
|
||||||
|
ResolveLookAtScratch* scratchpad;
|
||||||
|
};
|
||||||
|
|
||||||
|
/* Atom 0 in the bundle: input_and_sub. Stages C-side inputs into the scratchpad and computes fwd = target - eye.
|
||||||
|
* Staging work:
|
||||||
|
* * Stage eye.x/y/z → scratch (for atom 6's translation column)
|
||||||
|
* * Stage up_in.x/y/z → scratch (for atom 2's outer-product operand)
|
||||||
|
* * Compute fwd = target - eye, store fwd.x/y/z → scratch+0/+4/+8 (for atom 1)
|
||||||
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
|
* r_target_ptr : R_T0
|
||||||
|
* r_eye_ptr : R_T1
|
||||||
|
* r_up_in_ptr : R_T2
|
||||||
|
* r_scratch : R_T4 (R_ResolveScratch; wave-context carrier)
|
||||||
|
* r_tmp0 : R_T3 (stage eye/up_in + load eye.y)
|
||||||
|
* r_tmp1 : R_T5 (stage eye/up_in + load eye.z)
|
||||||
|
* r_tmp2 : R_T6 (stage eye/up_in + load target.x)
|
||||||
|
* r_tmp3 : R_T7 (stage eye/up_in + load target.y)
|
||||||
|
* R_AT : hardcoded (load eye.y / eye.z / target.z)
|
||||||
|
* R_V0 : hardcoded (load eye.z / target.z)
|
||||||
|
* Pool cost: 8 GPRs + R_T4 (carrier) + R_AT + R_V0 (hardcoded) = 11 GPRs.
|
||||||
|
*/
|
||||||
|
internal MipsAtom* resolve_look_at__input_and_sub_proc(AtomArena_R aa,
|
||||||
|
// TODO(Ed): We can resolve scratch at anytime its fixed to a specific address.
|
||||||
|
U4 r_scratch
|
||||||
|
, U4 r_target_ptr,U4 r_eye_ptr, U4 r_up_in_ptr
|
||||||
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2, U4 r_tmp3
|
||||||
|
) MipsAtom_Proc_(aa, {
|
||||||
|
load_word(r_target_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,target)),
|
||||||
|
load_word(r_eye_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,eye)),
|
||||||
|
load_word(r_up_in_ptr, R_TapePtr, O_(Binds_ResolveLookAtSub,up_in)),
|
||||||
|
load_word(r_scratch, R_TapePtr, O_(Binds_ResolveLookAtSub,scratchpad)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtSub)),
|
||||||
|
|
||||||
|
// Stage eye.x/y/z into the scratchpad (atom 6 reads these for the translation column).
|
||||||
|
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_eye_ptr, 0),
|
||||||
|
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||||
|
|
||||||
|
/* Stage up_in.x/y/z into the scratchpad. */
|
||||||
|
mac_load_p3s4( r_tmp0, r_tmp1, r_tmp2, r_up_in_ptr, 0),
|
||||||
|
mac_store_p3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,up_in)),
|
||||||
|
|
||||||
|
/* Compute fwd = target - eye. */
|
||||||
|
mac_load_p3s4(r_tmp0, r_tmp1, r_tmp2, r_target_ptr, 0),
|
||||||
|
mac_load_p3s4(r_tmp3, R_AT, R_V0, r_eye_ptr, 0),
|
||||||
|
mac_sub_v3s4(
|
||||||
|
r_tmp0, r_tmp1, r_tmp2,
|
||||||
|
r_tmp3, R_AT, R_V0),
|
||||||
|
mac_store_v3s4(r_tmp0, r_tmp1, r_tmp2, r_scratch, O_(ResolveLookAtScratch,fwd)),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Atom 2: cross uz × up_in → right. */
|
||||||
|
internal MipsAtom* resolve_look_at__cross_uz_up_in_to_right_proc(AtomArena_R aa, U4 r_scratch
|
||||||
|
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||||
|
, U4 r_d /* load b.x */
|
||||||
|
, U4 r_f, U4 r_g, U4 r_h /* r_f = &right (out ptr), r_g = &uz, r_h = &up_in */
|
||||||
|
) MipsAtom_Proc_(aa, {
|
||||||
|
/* FIX: build packed RT22+RT33 with proper sign extension. */
|
||||||
|
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||||
|
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,up_in)), /* r_h = &up_in */
|
||||||
|
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,right)), /* r_f = &right (out) */
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||||
|
load_word(r_a, r_g, O_(V3_S4,x)),
|
||||||
|
load_word(r_b, r_g, O_(V3_S4,y)),
|
||||||
|
load_word(r_c, r_g, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Load b (up_in).x/y/z into r_d + R_AT/R_V0 (R_AT/R_V0 are hardcoded scratch). */
|
||||||
|
load_word(r_d, r_h, O_(V3_S4,x)),
|
||||||
|
load_word(R_AT, r_h, O_(V3_S4,y)),
|
||||||
|
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Save the two RT control-register slots OP will clobber. We reuse
|
||||||
|
* r_g/r_h (scratch pointers, no longer needed) as the save targets. */
|
||||||
|
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 r0 (RT11|RT12) */
|
||||||
|
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 r4 (RT22|RT33) */
|
||||||
|
|
||||||
|
/* Load uz.x/uz.y/uz.z into COP2 control registers.
|
||||||
|
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
||||||
|
* RT22 is in BOTH $2.high AND $4.low (shared bit position). OP reads from $2.high.
|
||||||
|
* So set RT22 via ctc2 r_b, $2 (sets $2.high = a.y.high = RT22, $2.low = a.y.low = RT13).
|
||||||
|
* Then set RT33 via ctc2 r_c, $4 (sets $4.high = a.z.high = RT33, $4.low = a.z.low).
|
||||||
|
* The $2 and $4 writes don't clobber each other (separate registers).
|
||||||
|
* The 2nd ctc2 DOES clobber $4.low (becomes a.z.low, NOT a.y.high), but since OP
|
||||||
|
* reads RT22 from $2.high (which the 2nd ctc2 doesn't touch), D2 is still a.y.high.
|
||||||
|
* This is libpsyx's OuterProduct12 convention EXACTLY. */
|
||||||
|
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = r_b = a.y. RT13=a.y.low, RT22=a.y.high. */
|
||||||
|
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = r_c = a.z. RT22=a.z.low, RT33=a.z.high. */
|
||||||
|
|
||||||
|
/* Load uz into the RT diagonal. */
|
||||||
|
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* D1 = RT11 = uz.x (low 16 of $0, sign-extended by OP). */
|
||||||
|
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
/* Load up_in into IR (the second operand for OP). */
|
||||||
|
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = up_in.x */
|
||||||
|
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = up_in.y */
|
||||||
|
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = up_in.z */
|
||||||
|
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
gte_cmdw_outer_product, /* OP: MAC1/2/3 = uz × up_in
|
||||||
|
* MAC1 = IR3*D2 - IR2*D3 = up_in.z*uz.y.high - up_in.y*uz.z.high
|
||||||
|
* MAC2 = IR1*D3 - IR3*D1 = up_in.x*uz.z.high - up_in.z*uz.x
|
||||||
|
* MAC3 = IR2*D1 - IR1*D2 = up_in.y*uz.x - up_in.x*uz.y.high
|
||||||
|
* For up_in = (0, -fp_one, 0):
|
||||||
|
* MAC1 = 0 - (-fp_one)*uz.z.high = fp_one*uz.z.high
|
||||||
|
* MAC2 = 0 - 0 = 0
|
||||||
|
* MAC3 = (-fp_one)*uz.x - 0 = -fp_one*uz.x */
|
||||||
|
|
||||||
|
/* Restore the RT slots we clobbered. */
|
||||||
|
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 r0 (RT11|RT12) */
|
||||||
|
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 r4 (RT22|RT33) */
|
||||||
|
|
||||||
|
/* mfc2 MAC1/2/3 → r_a/r_b/r_c (out.x/y/z). */
|
||||||
|
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_b, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||||
|
nop, /* MFC2 retirement */
|
||||||
|
|
||||||
|
/* Right-shift MAC by 12 to convert from GTE's S12.20 fixed-point scale back to libpsyx OuterProduct12 convention (S12.0, fp_one=4096=1<<12).
|
||||||
|
* Without this, MAC values (~16M for unit-vector cross products) overflow the GTE's 16-bit IR registers when atom 3 normalizes via mtc2. */
|
||||||
|
shift_aright(r_a, r_a, 12),
|
||||||
|
shift_aright(r_b, r_b, 12),
|
||||||
|
shift_aright(r_c, r_c, 12),
|
||||||
|
|
||||||
|
/* Store out.x/y/z to r_f (out ptr = scratch+32). */
|
||||||
|
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||||
|
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||||
|
store_word(r_c, r_f, O_(V3_S4,z)),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Atom 4: cross uz × ux → up. */
|
||||||
|
internal MipsAtom* resolve_look_at__cross_uz_ux_to_up_proc(AtomArena_R aa, U4 r_scratch
|
||||||
|
, U4 r_a, U4 r_b, U4 r_c /* load a.x/y/z; result out.x/y/z */
|
||||||
|
, U4 r_d /* load b.x */
|
||||||
|
, U4 r_f, U4 r_g, U4 r_h /* r_f = &up (out ptr), r_g = &uz, r_h = &ux */
|
||||||
|
) MipsAtom_Proc_(aa, {
|
||||||
|
/* Compute the three scratch pointers from r_scratch. */
|
||||||
|
add_si(r_g, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_g = &uz */
|
||||||
|
add_si(r_h, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_h = &ux */
|
||||||
|
add_si(r_f, r_scratch, O_(ResolveLookAtScratch,up)), /* r_f = &up (out) */
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Load a (uz).x/y/z into r_a/r_b/r_c. */
|
||||||
|
load_word(r_a, r_g, O_(V3_S4,x)),
|
||||||
|
load_word(r_b, r_g, O_(V3_S4,y)),
|
||||||
|
load_word(r_c, r_g, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Load b (ux).x/y/z into r_d + R_AT/R_V0. */
|
||||||
|
load_word(r_d, r_h, O_(V3_S4,x)),
|
||||||
|
load_word(R_AT, r_h, O_(V3_S4,y)),
|
||||||
|
load_word(R_V0, r_h, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* OP reads D1/D2/D3 from RT11/RT22/RT33 ($0/$2/$4), not V0/V1/V2.
|
||||||
|
* Mirror atom 1: cfc2 RT save, ctc2 RT diagonal from uz, mtc2 IR from ux,
|
||||||
|
* ctc2 RT restore. */
|
||||||
|
|
||||||
|
/* Save the two RT control-register slots OP will clobber (reusing
|
||||||
|
* r_g/r_h — they're no longer needed as scratch pointers). */
|
||||||
|
gte_mv_from_ctrl_r(r_g, gte_cr_RT11), /* r_g = C2 $0 (RT11|RT12) */
|
||||||
|
gte_mv_from_ctrl_r(r_h, gte_cr_RT22), /* r_h = C2 $4 (RT22|RT33) */
|
||||||
|
|
||||||
|
/* Load uz into the RT diagonal — same packing as atom 1.
|
||||||
|
* OP reads D1 = RT11 from $0.low, D2 = RT22 from $2.high, D3 = RT33 from $4.high.
|
||||||
|
* RT22 is shared between $2.high and $4.low — the ctc2 sequence to $2 then $4
|
||||||
|
* sets RT22 to uz.y.high (via $2), then to uz.z.low (via $4). OP reads
|
||||||
|
* RT22 from $2.high which the second ctc2 doesn't touch, so D2 stays uz.y.high.
|
||||||
|
* (This is libpsyx OuterProduct12 convention EXACTLY.) */
|
||||||
|
gte_mv_to_ctrl_r(r_b, gte_cr_RT13), /* $2 = uz.y. RT13=uz.y.low, RT22=uz.y.high. */
|
||||||
|
gte_mv_to_ctrl_r(r_c, gte_cr_RT22), /* $4 = uz.z. RT22=uz.z.low, RT33=uz.z.high. */
|
||||||
|
gte_mv_to_ctrl_r(r_a, gte_cr_RT11), /* $0 = uz.x. RT11=uz.x. */
|
||||||
|
nop2, /* CTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
/* Load ux into the IR registers (the second operand for OP). */
|
||||||
|
gte_mv_to_data_r(r_d, C2_IR1), /* IR1 = ux.x */
|
||||||
|
gte_mv_to_data_r(R_AT, C2_IR2), /* IR2 = ux.y */
|
||||||
|
gte_mv_to_data_r(R_V0, C2_IR3), /* IR3 = ux.z */
|
||||||
|
nop2, /* MTC2 retirement (CPU→COP2 2-slot delay) */
|
||||||
|
|
||||||
|
gte_cmdw_outer_product,
|
||||||
|
|
||||||
|
/* Restore the RT slots we clobbered. */
|
||||||
|
gte_mv_to_ctrl_r(r_g, gte_cr_RT11), /* restore C2 $0 (RT11|RT12) */
|
||||||
|
gte_mv_to_ctrl_r(r_h, gte_cr_RT22), /* restore C2 $4 (RT22|RT33) */
|
||||||
|
|
||||||
|
gte_mv_from_data_r(r_a, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_b, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_c, C2_MAC3),
|
||||||
|
nop,
|
||||||
|
/* Right-shift MAC by 12 to convert from GTE's S12.20 scale back to libpsyx
|
||||||
|
* OuterProduct12 convention (S12.0, fp_one=4096). See atom 1 for rationale. */
|
||||||
|
shift_aright(r_a, r_a, 12),
|
||||||
|
shift_aright(r_b, r_b, 12),
|
||||||
|
shift_aright(r_c, r_c, 12),
|
||||||
|
store_word(r_a, r_f, O_(V3_S4,x)),
|
||||||
|
store_word(r_b, r_f, O_(V3_S4,y)),
|
||||||
|
store_word(r_c, r_f, O_(V3_S4,z)),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
typedef Struct_(Binds_ResolveLookAtPopAndTrans) {
|
||||||
|
U4 look_at; /* U4 (MT3_S2S4* — destination matrix address) */
|
||||||
|
};
|
||||||
|
/* Atom 6 in the bundle: write look_at->m[][] from ux/uy/uz, then compute the translation column t[] = R * (-eye).
|
||||||
|
*
|
||||||
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
|
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
||||||
|
* r_pux : pointer to ux (offset O_(ResolveLookAtScratch,ux))
|
||||||
|
* r_puy : pointer to uy (offset O_(ResolveLookAtScratch,uy))
|
||||||
|
* r_puz : pointer to uz (offset O_(ResolveLookAtScratch,uz))
|
||||||
|
* r_peye : pointer to eye (offset O_(ResolveLookAtScratch,eye))
|
||||||
|
* r_tmp0/1/2 : atom-local scratch (load + MVMVA + store temps)
|
||||||
|
*
|
||||||
|
* 4 pointer regs (r_pux/r_puy/r_puz/r_peye) are DEDICATED — they hold the scratch addresses for the entire body.
|
||||||
|
* They are computed in-body via `add_si(r_px, r_scratch, O_(ResolveLookAtScratch, field))` so no tape-data pointer is needed.
|
||||||
|
*
|
||||||
|
* Struct layout (per duffle/math.h):
|
||||||
|
* MT3_S2S4 { A3x3_S2 m; A3_S4 t; } → m[][] is S2 packed (9 × 2 = 18 bytes at offset 0)
|
||||||
|
* t[0/1/2] is S4 (3 × 4 = 12 bytes at offset 18)
|
||||||
|
*
|
||||||
|
* Translation column: GTE MVMVA with the world rotation matrix pre-set
|
||||||
|
* (helper emits set_gte_world before the bundle, per the bundle design).
|
||||||
|
* MVMVA computes R * pos (with cv=0/mx=0/sf=0/v=0); MAC1/2/3 = R * (-eye).
|
||||||
|
* Pool cost: r_look_at (1) + r_scratch (R_T4 carrier) + 4 ptr regs + 3 tmp regs = 9 GPRs.
|
||||||
|
*/
|
||||||
|
internal MipsAtom* resolve_look_at__populate_proc(AtomArena_R aa
|
||||||
|
, U4 r_look_at
|
||||||
|
, U4 r_scratch
|
||||||
|
, U4 r_pux, U4 r_puy, U4 r_puz
|
||||||
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||||
|
) MipsAtom_Proc_(aa, {
|
||||||
|
/* Pop look_at* (the matrix output) — advance R_TapePtr by 4 bytes. */
|
||||||
|
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||||
|
|
||||||
|
/* Compute the 3 scratch pointers in their dedicated GPRs (eye isn't needed by 6a — 6b reads it). */
|
||||||
|
add_si(r_pux, r_scratch, O_(ResolveLookAtScratch,ux)), /* r_pux = &ux */
|
||||||
|
add_si(r_puy, r_scratch, O_(ResolveLookAtScratch,uy)), /* r_puy = &uy */
|
||||||
|
add_si(r_puz, r_scratch, O_(ResolveLookAtScratch,uz)), /* r_puz = &uz */
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* ── m[0] = (S2)ux ── */
|
||||||
|
load_word(r_tmp0, r_pux, O_(V3_S4,x)),
|
||||||
|
load_word(r_tmp1, r_pux, O_(V3_S4,y)),
|
||||||
|
load_word(r_tmp2, r_pux, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[0][0])),
|
||||||
|
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[0][1])),
|
||||||
|
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[0][2])),
|
||||||
|
|
||||||
|
/* ── m[1] = (S2)uy ── */
|
||||||
|
load_word(r_tmp0, r_puy, O_(V3_S4,x)),
|
||||||
|
load_word(r_tmp1, r_puy, O_(V3_S4,y)),
|
||||||
|
load_word(r_tmp2, r_puy, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[1][0])),
|
||||||
|
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[1][1])),
|
||||||
|
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[1][2])),
|
||||||
|
|
||||||
|
/* ── m[2] = (S2)uz ── */
|
||||||
|
load_word(r_tmp0, r_puz, O_(V3_S4,x)),
|
||||||
|
load_word(r_tmp1, r_puz, O_(V3_S4,y)),
|
||||||
|
load_word(r_tmp2, r_puz, O_(V3_S4,z)),
|
||||||
|
nop,
|
||||||
|
store_half(r_tmp0, r_look_at, O_(MT3_S2S4,m[2][0])),
|
||||||
|
store_half(r_tmp1, r_look_at, O_(MT3_S2S4,m[2][1])),
|
||||||
|
store_half(r_tmp2, r_look_at, O_(MT3_S2S4,m[2][2])),
|
||||||
|
|
||||||
|
/* Zero t[0..2] — atom 6c writes the final values here. */
|
||||||
|
store_word(R_0, r_look_at, O_(MT3_S2S4,t[0])),
|
||||||
|
store_word(R_0, r_look_at, O_(MT3_S2S4,t[1])),
|
||||||
|
store_word(R_0, r_look_at, O_(MT3_S2S4,t[2])),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Atom 6b in the bundle: matrix-vector product off = R * (-eye) >> 12.
|
||||||
|
* Uses RTPS with V0 loaded from scratch via lwc2. The RT matrix is
|
||||||
|
* pre-loaded by atom 6a.5 (resolve_look_at__load_rt).
|
||||||
|
* Stores off to scratch+96 (overwriting the packed pos).
|
||||||
|
*
|
||||||
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
|
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
||||||
|
* r_peye : pointer to eye (slot +96, reused as off destination)
|
||||||
|
* r_tmp0/1/2: -eye + GTE transfer scratch
|
||||||
|
*
|
||||||
|
* Pool cost: r_scratch (carrier) + 1 ptr reg + 3 tmp regs = 5 GPRs.
|
||||||
|
*/
|
||||||
|
internal MipsAtom* resolve_look_at__matrix_vector_proc(AtomArena_R aa
|
||||||
|
, U4 r_scratch
|
||||||
|
, U4 r_peye
|
||||||
|
, U4 r_look_at
|
||||||
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||||
|
) MipsAtom_Proc_(aa, {
|
||||||
|
/* === EXACT C11 ApplyMatrixLV replication ===
|
||||||
|
* The C11 does:
|
||||||
|
* 1. ctc2 RT matrix (5 ctc2s to C2[0..4])
|
||||||
|
* 2. lw v.x/y/z from memory
|
||||||
|
* 3. S15 decomposition (negu + sra 15 + negu + andi 0x7FFF + negu)
|
||||||
|
* 4. mtc2 HIGH bits to IR1/2/3, nop, MVMVA pass1 (sf=0, mx=0, v=3, cv=3)
|
||||||
|
* 5. mfc2 MACs
|
||||||
|
* 6. mtc2 LOW bits to IR1/2/3, nop, MVMVA pass2 (sf=1, mx=0, v=3, cv=3)
|
||||||
|
* 7. mfc2 MACs
|
||||||
|
* 8. Combine: (pass1 << 3) + pass2
|
||||||
|
*
|
||||||
|
* For S16-fitting pos (|pos| < 32768), pos >> 15 = 0, so pass1 = 0.
|
||||||
|
* The combine simplifies: result = 0 + pass2 = pass2.
|
||||||
|
* So we skip the S15 decomposition and just do pass 2 directly.
|
||||||
|
* We still use v=3 (IR input) and mx=0 (RT matrix) like the C11. */
|
||||||
|
|
||||||
|
/* Pop look_at* from tape. */
|
||||||
|
load_word(r_look_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||||
|
|
||||||
|
/* r_peye = &eye (slot +96, reused as off destination). */
|
||||||
|
add_si(r_peye, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* === Load RT matrix from look_at into C2[0..4] via ctc2 ===
|
||||||
|
* Exact s ame sequence as set_gte_mt3s2s4 / C11's ApplyMatrixLV. */
|
||||||
|
load_word( r_tmp0, r_look_at, 0), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT11),
|
||||||
|
load_word( r_tmp0, r_look_at, 4), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT12),
|
||||||
|
load_word( r_tmp0, r_look_at, 8), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT13),
|
||||||
|
load_word( r_tmp0, r_look_at, 12), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT21),
|
||||||
|
load_half_u(r_tmp0, r_look_at, 16), nop, gte_mv_to_ctrl_r(r_tmp0, gte_cr_RT22),
|
||||||
|
nop2, /* CTC2 retirement (2 slots × 5 ctc2s) */
|
||||||
|
|
||||||
|
/* Load pos = -eye after the matrix load releases r_tmp0. */
|
||||||
|
load_word(r_tmp0, r_peye, O_(P3_S4,x)),
|
||||||
|
load_word(r_tmp1, r_peye, O_(P3_S4,y)),
|
||||||
|
load_word(r_tmp2, r_peye, O_(P3_S4,z)),
|
||||||
|
nop,
|
||||||
|
sub_u(r_tmp0, R_0, r_tmp0), /* pos.x = -eye.x */
|
||||||
|
sub_u(r_tmp1, R_0, r_tmp1),
|
||||||
|
sub_u(r_tmp2, R_0, r_tmp2),
|
||||||
|
|
||||||
|
/* === mtc2 pos (as S16) to IR1/2/3 ===
|
||||||
|
* The GTE takes low 16 bits. pos fits in S16. For negative pos, the
|
||||||
|
* 32-bit sign-extended value's low 16 bits = correct S16. */
|
||||||
|
/* Mask pos to 16 bits to be safe. For S16-fitting pos, pos & 0xFFFF
|
||||||
|
* gives the correct S16 value (sign bit preserved). */
|
||||||
|
/* r_tmp0/1/2 already have pos values. */
|
||||||
|
gte_mv_to_data_r(r_tmp0, C2_IR1),
|
||||||
|
gte_mv_to_data_r(r_tmp1, C2_IR2),
|
||||||
|
gte_mv_to_data_r(r_tmp2, C2_IR3),
|
||||||
|
nop2, /* MTC2 retirement (2 slots) */
|
||||||
|
|
||||||
|
/* === MVMVA pass 2 — C11 ApplyMatrixLV command ===
|
||||||
|
* sf=1, mx=0 (RT), v=3 (IR), cv=3. Reads RT × IR >> 12. */
|
||||||
|
gte_cmdw_mvmva_c11_pass2,
|
||||||
|
nop, /* GTE interlock */
|
||||||
|
|
||||||
|
/* === mfc2 MAC1/2/3 → r_tmp0/1/2 === */
|
||||||
|
gte_mv_from_data_r(r_tmp0, C2_MAC1),
|
||||||
|
gte_mv_from_data_r(r_tmp1, C2_MAC2),
|
||||||
|
gte_mv_from_data_r(r_tmp2, C2_MAC3),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* === Store off → scratch+96 (overwriting pos) === */
|
||||||
|
store_word(r_tmp0, r_peye, O_(V3_S4,x)),
|
||||||
|
store_word(r_tmp1, r_peye, O_(V3_S4,y)),
|
||||||
|
store_word(r_tmp2, r_peye, O_(V3_S4,z)),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
/* Atom 6c in the bundle: copy scratch+96 (off, written by atom 6b) → look_at->t[].
|
||||||
|
* Uses mac_trans_matrix component (m->t = v, libgte TransMatrix semantics = struct copy).
|
||||||
|
*
|
||||||
|
* GPR codes (assigned by resolve_look_at_init):
|
||||||
|
* r_look_at : MT3_S2S4* (popped from tape; output matrix destination)
|
||||||
|
* r_scratch : R_ResolveScratch (R_T4) — scratch base
|
||||||
|
* r_off_ptr : pointer to off (= &scratch.eye, reused slot)
|
||||||
|
* r_tmp0 : transfer reg for mac_trans_matrix
|
||||||
|
*
|
||||||
|
* Pool cost: r_look_at (1) + r_scratch (carrier) + r_off_ptr + 1 clobber = 4 GPRs.
|
||||||
|
*/
|
||||||
|
I_ MipsAtom* resolve_look_at__trans_matrix_proc(AtomArena_R aa
|
||||||
|
, U4 r_look_at, U4 r_scratch, U4 r_off_ptr
|
||||||
|
, U4 r_tmp0, U4 r_tmp1, U4 r_tmp2
|
||||||
|
) MipsAtom_Proc_(aa, {
|
||||||
|
/* Pop look_at* from tape. */
|
||||||
|
// load_word(r_Vlook_at, R_TapePtr, O_(Binds_ResolveLookAtPopAndTrans,look_at)),
|
||||||
|
// add_ui_self( R_TapePtr, S_(Binds_ResolveLookAtPopAndTrans)),
|
||||||
|
|
||||||
|
/* r_off_ptr = &off (= &scratch.eye since atom 6b overwrote eye with off). */
|
||||||
|
add_si(r_off_ptr, r_scratch, O_(ResolveLookAtScratch,eye)),
|
||||||
|
nop,
|
||||||
|
|
||||||
|
/* Copy off → look_at.t[] (mac_trans_matrix: m->t = v). */
|
||||||
|
mac_trans_mt3s3s4(r_look_at, r_off_ptr, r_tmp0, r_tmp1, r_tmp2),
|
||||||
|
|
||||||
|
mac_yield()
|
||||||
|
})
|
||||||
|
|
||||||
|
#pragma endregion Atom Procs
|
||||||
|
|
||||||
|
#pragma region Baked Atoms
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_ScreenX = R_T5 atom_reg atom_type(U2),
|
||||||
|
R_ScreenY = R_T6 atom_reg atom_type(U2),
|
||||||
|
R_ScreenBuf = R_T7 atom_reg, /* Caller-pinned: & smem.screen_buf */
|
||||||
|
#define R_ScreenBuf_Code R_T7_Code
|
||||||
|
};
|
||||||
|
//screen_env_init. Mirrors the libpsyx's SetDefDispEnv + SetDefDrawEnv + the manual enable_auto_clear / initial_bg_color writes.
|
||||||
|
internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
||||||
|
, atom_reads(R_T0, R_ScreenX, R_ScreenY, R_ScreenBuf)
|
||||||
|
, atom_writes(R_T0, R_ScreenX, R_ScreenY)
|
||||||
|
) {
|
||||||
|
/* display[0] = (0, 0, 320, 240); rest of struct zeroed. */
|
||||||
|
add_ui(R_ScreenX, R_0, ScreenRes_X), add_ui(R_ScreenY, R_0, ScreenRes_Y),
|
||||||
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area.width) + O_(DoubleBuffer,display[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[0])),
|
||||||
|
|
||||||
|
/* display[1] = (0, 240, 320, 240); rest of struct zeroed. */
|
||||||
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DisplayEnv,display_area) + O_(DoubleBuffer,display[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,screen) + O_(DoubleBuffer,display[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + O_(DoubleBuffer,display[1])),
|
||||||
|
|
||||||
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + O_(DoubleBuffer,draw[0])), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
|
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + O_(DoubleBuffer,draw[0])), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0].texture_window = (0, 0, 0, 0); two word-zeroes cover the full 8-byte tw field. */
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[0])),
|
||||||
|
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,drawing_offset[0].x) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.x) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||||
|
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||||
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0] control bytes: flag_dither=1, flag_draw_on_display=1 (the dfe bit per psx-spx; libpsyx sets it via `SetDefDrawEnv`'s conditional at C11_only.elf:0x80012728), enable_auto_clear=1. Each byte is named;
|
||||||
|
* the previous `store_word(R_0, ..., +20)` overwrote all four with zero. */
|
||||||
|
add_ui(R_T0, R_0, 1),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[0])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_dither) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,flag_draw_on_display) + O_(DoubleBuffer,draw[1])),
|
||||||
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||||
|
add_ui(R_T0, R_0, 7),
|
||||||
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[0])),
|
||||||
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + O_(DoubleBuffer,draw[1])),
|
||||||
|
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
/* gp_screen_init's GPR setup. Tests the mixed user-pinning + auto-reg pattern:
|
||||||
|
* - R_IO_BaseAddr = R_T4 (user-pinned via atom_reg; pre-existing)
|
||||||
|
* - R_GP1_Offset = R_T2 (user-pinned via atom_reg; NEW -- for GPIO_PORT1_OFFSET)
|
||||||
|
* - R_ScreenX = R_T5 (user-pinned via atom_reg; used as a transfer and GTE setup reg)
|
||||||
|
* - R_GpTmp = auto-allocated by the lua pass and used for several GPU transfers;
|
||||||
|
* the C preprocessor resolves it to the chosen free pool GPR.
|
||||||
|
*
|
||||||
|
* For gp_screen_init, the auto-reg pool exclusions are:
|
||||||
|
* user_pinned (from the corpus register_alias_registry) : R_T0..R_T7 (all 8 user-pinned across hello_camera.atom.c)
|
||||||
|
* body-parsed physical registers : aliases resolve through the registry;
|
||||||
|
* the body uses R_ScreenX, not raw R_T5
|
||||||
|
* source_pool after both subtractions : {R_V0, R_V1} only
|
||||||
|
* R_GpTmp gets R_V0 (the first-fit choice). Its repeated GPU-transfer use proves that the
|
||||||
|
* auto-reg allocation is active while the R_ScreenX references prove the pinned alias is used.
|
||||||
|
* R_TapePtr (R_T9), R_AtomJmp (R_T8), R_AT are excluded from the POOL by construction in
|
||||||
|
* passes/auto_reg.lua -- see the "obvious exclusions" comment block at the top of that file.
|
||||||
|
*/
|
||||||
|
enum {
|
||||||
|
R_IO_BaseAddr = R_T4 atom_reg, /* Caller-pinned: IO_BASE_ADDR = 0x1F800000 */
|
||||||
|
R_GP1_Offset = R_T2 atom_reg, /* Caller-pinned: GPIO_PORT1_OFFSET = 0x10 */
|
||||||
|
atom_auto_reg(gp_screen_init, R_GpTmp), /* Auto-allocated scratch; resolved to a free pool GPR by the lua pass. C-preprocessor expands to R_GpTmp = R_GpTmp_Code with an atom_auto_reg trailing comment. */
|
||||||
|
#define R_IO_BaseAddr_Code R_T4_Code
|
||||||
|
#define R_GP1_Offset_Code R_T2_Code
|
||||||
|
};
|
||||||
|
internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads(R_IO_BaseAddr)) {
|
||||||
|
store_word(R_0, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(00h) Reset */
|
||||||
|
mac_gcmd_push(gp1_word_ResetCmdBuffer(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(01h) ClearFIFO; uses pinned R_ScreenX as the transfer reg. */
|
||||||
|
mac_gcmd_push(gp1_word_AcknowledgeIRQ(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(02h) AckIRQ; uses pinned R_ScreenX as the transfer reg. */
|
||||||
|
mac_gcmd_push(gp1_word_DisplayOn(), R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(03h) Display ON; uses pinned R_ScreenX as the transfer reg. */
|
||||||
|
mac_gcmd_push(gp1_word_dma_to_gpu(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(04h) DMADirection=2 (CPU->GPU). libpsyx's per-frame PutDrawEnv/DrawOTag use DMA2; without this the DMA queue never drains. Uses auto-allocated R_GpTmp. */
|
||||||
|
mac_gcmd_push(gp1_word_StartDisplayArea(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* GP1(05h) StartDisplayArea (X=0, Y=0); uses auto-allocated R_GpTmp. */
|
||||||
|
|
||||||
|
/* GP1: DisplayMode + Display Ranges. */
|
||||||
|
mac_gcmd_push(gp1_word_display_mode_320x240_15bit_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_gcmd_push(gp1_word_horizontal_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
mac_gcmd_push(gp1_word_vertical_range_ntsc, R_ScreenX, R_IO_BaseAddr, GPIO_PORT1_OFFSET),
|
||||||
|
|
||||||
|
/* GTE: SetGeomOffset (OFX, OFY) — ScreenRes_CenterX, ScreenRes_CenterY. */
|
||||||
|
load_upper_i(R_ScreenX, ScreenRes_CenterX), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFX_Code),
|
||||||
|
load_upper_i(R_ScreenX, ScreenRes_CenterY), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_OFY_Code),
|
||||||
|
|
||||||
|
/* GTE: SetGeomScreen (H) — CR26 (per PSX-SPX / libpsyx), value is the raw projection-plane distance, NOT shifted. */
|
||||||
|
add_ui(R_ScreenX, R_0, ScreenZ), gte_mv_to_ctrl_r(R_ScreenX, gte_cr_H_Code),
|
||||||
|
|
||||||
|
/* GP1: DisplayEnable — bit 0 = 0 (Display ON). */
|
||||||
|
mac_gcmd_push(gp1_word_DisplayOn(), R_GpTmp, R_IO_BaseAddr, GPIO_PORT1_OFFSET), /* Uses auto-allocated R_GpTmp. */
|
||||||
|
mac_yield(),
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_PadApplyInput) {
|
||||||
|
PadState* state;
|
||||||
|
V3_S2* cube_rot;
|
||||||
|
V3_S2* floor_rot;
|
||||||
|
};
|
||||||
|
enum {
|
||||||
|
R_PadStateT5 = R_T5 atom_reg,
|
||||||
|
R_CubeRot = R_T1 atom_reg,
|
||||||
|
R_FloorRot = R_T2 atom_reg,
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_input_cube_rotation) atom_info(atom_bind(Binds_PadApplyInput)
|
||||||
|
, atom_reads(R_T0, R_CubeRot, R_FloorRot, R_T3, R_T4, R_PadStateT5, R_TapePtr)
|
||||||
|
, atom_writes( R_CubeRot, R_FloorRot)
|
||||||
|
) {
|
||||||
|
/* Pop Binds from tape (state, cube_rot, floor_rot) */
|
||||||
|
load_word(R_PadStateT5, R_TapePtr, O_(Binds_PadApplyInput,state)),
|
||||||
|
load_word(R_CubeRot, R_TapePtr, O_(Binds_PadApplyInput,cube_rot)),
|
||||||
|
load_word(R_FloorRot, R_TapePtr, O_(Binds_PadApplyInput,floor_rot)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_PadApplyInput)),
|
||||||
|
|
||||||
|
/* Load pad[0].buttons into R_T0. */
|
||||||
|
load_word(R_T0, R_PadStateT5, O_(PadState,buttons)), nop,
|
||||||
|
// Note(Ed): Potential op with delay slot?
|
||||||
|
|
||||||
|
/* D-pad Left: cube_rot.y += 30, floor_rot.y += 5. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(dpad_left, exit_dpad_left)),
|
||||||
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||||
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
add_si( R_T4, R_T4, 30),
|
||||||
|
add_si( R_T3, R_T3, 5),
|
||||||
|
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
atom_label(exit_dpad_left)
|
||||||
|
|
||||||
|
/* D-pad Right: cube_rot.y -= 30, floor_rot.y -= 5. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(dpad_right, exit_dpad_right)),
|
||||||
|
load_half( R_T4, R_CubeRot, O_(V3_S2,y)), /* BD-slot */
|
||||||
|
load_half( R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
add_si( R_T4, R_T4, -30),
|
||||||
|
add_si( R_T3, R_T3, -5),
|
||||||
|
store_half(R_T4, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
store_half(R_T3, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
atom_label(exit_dpad_right)
|
||||||
|
|
||||||
|
/* Analog left-stick X: dead zone 0x70..0x90.
|
||||||
|
* Cube delta = (0x80 - left_x) >> 2; floor delta = (0x80 - left_x) >> 5. */
|
||||||
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)),
|
||||||
|
|
||||||
|
/* Dead-zone check: skip analog if left_x in [0x70, 0x90] inclusive. Outside dead zone on LOW side: left_x < 0x70 (strictly).
|
||||||
|
* set_lt_u(R_T4, R_T3, R_T4=0x70) → R_T4 = (left_x < 0x70) ? 1 : 0. */
|
||||||
|
add_ui(R_T4, R_0, PadDeadZone_HighBound), set_lt_u(R_T4, R_T3, R_T4), branch_ne(R_T4, R_0, atom_offset(dead_zone_low_check, dead_low_active)),
|
||||||
|
add_ui(R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_low_active */
|
||||||
|
|
||||||
|
atom_label(dead_check_upper)
|
||||||
|
/* left_x >= 0x70 → check upper bound. */
|
||||||
|
load_byte_u(R_T3, R_PadStateT5, O_(PadState,left.x)), /* reload */
|
||||||
|
add_ui( R_T4, R_0, PadDeadZone_HighBound),
|
||||||
|
|
||||||
|
/* R_T4 = (0x90 < left_x) ? 1 : 0 → (left_x > 0x90) ? 1 : 0 */
|
||||||
|
set_lt_u(R_T4, R_T4, R_T3), branch_ne(R_T4, R_0, atom_offset(dead_zone_high_check, dead_high_active)),
|
||||||
|
add_ui( R_T4, R_0, PadDeadZone_Center), /* BD-slot: pre-load 0x80 for dead_high_active */
|
||||||
|
jump_rel(atom_offset(dead_zone_skip, exit_stick)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(dead_low_active)
|
||||||
|
/* R_T3 = left_x (from line 632 lbu; not clobbered between dead_zone_low_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
|
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||||
|
* R_T4 = 0x80 from the BD-slot of `dead_zone_low_check`'s branch_ne. */
|
||||||
|
sub_u( R_T3, R_T4, R_T3), /* R_T3 = 0x80 - left_x */
|
||||||
|
/* delta = 0x80 - left_x (positive). */
|
||||||
|
|
||||||
|
/* R_T4 = cube_delta */
|
||||||
|
shift_aright(R_T4, R_T3, 2),
|
||||||
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
/* R_T4 = floor_delta — moved into the load-delay slot of the floor load below (fills the 1-instruction gap;
|
||||||
|
* doesn't read R_T0; R_T4 settles by the subsequent add_u). */
|
||||||
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
shift_aright(R_T4, R_T3, 5),
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
jump_rel(atom_offset(end_low, exit_stick)),
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(dead_high_active)
|
||||||
|
/* R_T3 = left_x (from line 641 lbu in dead_check_upper; not clobbered between dead_zone_high_check branch + its BD-slot `add_ui R_T4, 0x80`).
|
||||||
|
* The earlier `load_byte_u(R_T3, ...)` reload was redundant and introduced a load-use hazard on the next `sub_u`.
|
||||||
|
* R_T4 = 0x80 from the BD-slot of `dead_zone_high_check`'s branch_ne. */
|
||||||
|
sub_u( R_T3, R_T4, R_T3),
|
||||||
|
/* delta = 0x80 - left_x (signed negative). */
|
||||||
|
|
||||||
|
shift_aright(R_T4, R_T3, 2), /* R_T4 = cube_delta (signed) */
|
||||||
|
load_half( R_T0, R_CubeRot, O_(V3_S2,y)), nop,
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_CubeRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
/* R_T4 = floor_delta (signed) — moved into the load-delay slot of the floor load below. */
|
||||||
|
load_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
shift_aright(R_T4, R_T3, 5),
|
||||||
|
add_u( R_T0, R_T0, R_T4),
|
||||||
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
|
atom_label(exit_stick)
|
||||||
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_Cam = R_T4 atom_reg,
|
||||||
|
R_CamPadState = R_T5 atom_reg,
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_PadInputCam) {
|
||||||
|
PadState* state;
|
||||||
|
Camera* cam;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(pad_input_cam) atom_info(atom_bind(Binds_PadInputCam)
|
||||||
|
, atom_reads( R_Cam, R_CamPadState, R_TapePtr)
|
||||||
|
, atom_writes(R_Cam)
|
||||||
|
) {
|
||||||
|
/* Bind pop: state → R_CamPadState (R_T5), cam → R_Cam (R_T4), advance R_TapePtr by 8. */
|
||||||
|
load_word(R_CamPadState, R_TapePtr, O_(Binds_PadInputCam,state)),
|
||||||
|
load_word(R_Cam, R_TapePtr, O_(Binds_PadInputCam,cam)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_PadInputCam)),
|
||||||
|
|
||||||
|
/* Load pad[0].buttons into R_T0; nop fills the load-delay slot. */
|
||||||
|
load_word(R_T0, R_CamPadState, O_(PadState,buttons)),
|
||||||
|
load_word(R_T1, R_Cam, O_(Camera,pos.x)), // BD-Slot.
|
||||||
|
|
||||||
|
// D-pad Left → cam.pos.x -= 50. and_i fulfills BD-slot for load on R_Cam.
|
||||||
|
and_i(R_T3, R_T0, Pad_Left), branch_le_zero(R_T3, atom_offset(left_x, exit_left_x)), mac_yield_load(),
|
||||||
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
|
atom_label(exit_left_x)
|
||||||
|
|
||||||
|
/* D-pad Right → cam.pos.x += 50. Reuses R_T1 from Left. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Right), branch_le_zero(R_T3, atom_offset(right_x, exit_right_x)), nop,
|
||||||
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.x)),
|
||||||
|
atom_label(exit_right_x)
|
||||||
|
|
||||||
|
/* D-pad Up → cam.pos.y -= 50. Load pos.y BEFORE the andi. */
|
||||||
|
load_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
|
and_i(R_T3, R_T0, Pad_Up), branch_le_zero(R_T3, atom_offset(up_y, exit_up_y)), nop,
|
||||||
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
|
atom_label(exit_up_y)
|
||||||
|
|
||||||
|
/* D-pad Down → cam.pos.y += 50. Reuses R_T1 from Up. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Down), branch_le_zero(R_T3, atom_offset(down_y, exit_down_y)), nop,
|
||||||
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.y)),
|
||||||
|
atom_label(exit_down_y)
|
||||||
|
|
||||||
|
/* D-pad Cross → cam.pos.z -= 50. Load pos.z BEFORE the andi. */
|
||||||
|
load_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
|
and_i(R_T3, R_T0, Pad_Cross), branch_le_zero(R_T3, atom_offset(cross_z, exit_cross_z)), nop,
|
||||||
|
add_si(R_T1, R_T1, -50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
|
atom_label(exit_cross_z)
|
||||||
|
|
||||||
|
/* D-pad Circle → cam.pos.z += 50. Reuses R_T1 from Cross. */
|
||||||
|
and_i(R_T3, R_T0, Pad_Circle), branch_le_zero(R_T3, atom_offset(circle_z, exit_circle_z)), nop,
|
||||||
|
add_si(R_T1, R_T1, 50), store_word(R_T1, R_Cam, O_(Camera,pos.z)),
|
||||||
|
atom_label(exit_circle_z)
|
||||||
|
|
||||||
|
mac_yield_tail(),
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* Output cursor (primitive buffer) */
|
||||||
|
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||||
|
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||||
|
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||||
|
#define R_PrimCursor_Code R_T7_Code
|
||||||
|
#define R_FaceCursor_Code R_T4_Code
|
||||||
|
#define R_VertBase_Code R_T5_Code
|
||||||
|
#define R_OtBase_Code R_T6_Code
|
||||||
|
};
|
||||||
|
typedef Struct_(Binds_CubeTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V4_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal MipsAtom_(rbind_cube_g4_face) atom_info(atom_bind(Binds_CubeTri), atom_phase(cube_g4)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_CubeTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_CubeTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_CubeTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_CubeTri,OtBase)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_CubeTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// cube_g4_face — Draw one cube face (Gouraud-shaded quad) via the GTE tape pipeline
|
||||||
|
internal
|
||||||
|
MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
||||||
|
atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase),
|
||||||
|
atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
){
|
||||||
|
load_half_u(R_T0, R_FaceCursor, 0 * S_(S2)),
|
||||||
|
load_half_u(R_T1, R_FaceCursor, 1 * S_(S2)),
|
||||||
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
|
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||||
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||||
|
/* BD-slot: Write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||||
|
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||||
|
* harmless because the OT entry that points to this prim is created later. */
|
||||||
|
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||||
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
|
mac_gte_store_g4_p012(R_PrimCursor),
|
||||||
|
gte_cmdw_rotate_translate_perspective_single,
|
||||||
|
mac_gte_store_g4_p3(R_PrimCursor),
|
||||||
|
|
||||||
|
gte_cmdw_avg_sort_z4,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||||
|
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_G4)),
|
||||||
|
mac_format_g4_color(R_PrimCursor,
|
||||||
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
|
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||||
|
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||||
|
/* c3 green */ 0x00, 0xFF, 0x00),
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(cull)
|
||||||
|
|
||||||
|
atom_label(cube_g4_face_exit)
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_G4)), /* 9 words = Poly_G4 */
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* 4 × S2 = 8 bytes */
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_FloorTri) {
|
||||||
|
U4 PrimCursor;
|
||||||
|
V3_S2* FaceCursor;
|
||||||
|
V3_S2* VertBase;
|
||||||
|
U4* OtBase;
|
||||||
|
};
|
||||||
|
internal
|
||||||
|
MipsAtom_(rbind_floor_f3_face) atom_info(atom_bind(Binds_FloorTri), atom_phase(floor_f3)
|
||||||
|
, atom_reads(R_TapePtr)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase, R_TapePtr)
|
||||||
|
){
|
||||||
|
/* Pop 4 arguments from the tape directly into the workspace registers */
|
||||||
|
load_word(R_PrimCursor, R_TapePtr, O_(Binds_FloorTri,PrimCursor)),
|
||||||
|
load_word(R_FaceCursor, R_TapePtr, O_(Binds_FloorTri,FaceCursor)),
|
||||||
|
load_word(R_VertBase, R_TapePtr, O_(Binds_FloorTri,VertBase)),
|
||||||
|
load_word(R_OtBase, R_TapePtr, O_(Binds_FloorTri,OtBase)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_FloorTri)),
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
// atom_dbg_skip
|
||||||
|
internal
|
||||||
|
MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
||||||
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
|
) {
|
||||||
|
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||||
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
|
/* Culling (Branch forward if Backface) */
|
||||||
|
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||||
|
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||||
|
/* Format Primitive */
|
||||||
|
mac_gte_store_f3(R_PrimCursor),
|
||||||
|
|
||||||
|
/* Calculate Depth */
|
||||||
|
gte_avg_sort_z3,
|
||||||
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
|
/* Bounds Check OTZ < 2048 (Branch forward to skip insertion) */
|
||||||
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||||
|
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||||
|
mac_insert_ot_tag(R_OtBase, R_PrimCursor, S_(Poly_F3)), /* Insert into Ordering Table Linked List */
|
||||||
|
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||||
|
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||||
|
// end: branch(bounds_chk)
|
||||||
|
// end: branch(culling)
|
||||||
|
|
||||||
|
/* Advance Input Cursor & Yield (Both branch targets land here) */
|
||||||
|
atom_label(floor_f3_face_exit)
|
||||||
|
add_ui_self(R_FaceCursor, S_(S2) * 4), /* Advance Face Cursor (4 * S2 = 8 bytes) */
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Binds_SyncPrimitiveArena) { U4 used; U4 cursor; };
|
||||||
|
internal MipsAtom_(sync_primitive_arena) atom_info(atom_bind(Binds_SyncPrimitiveArena)
|
||||||
|
, atom_reads( R_TapePtr, R_PrimCursor)
|
||||||
|
, atom_writes(R_TapePtr)
|
||||||
|
){
|
||||||
|
load_word(R_AT, R_TapePtr, O_(Binds_SyncPrimitiveArena,used)),
|
||||||
|
load_word(R_T0, R_TapePtr, O_(Binds_SyncPrimitiveArena,cursor)),
|
||||||
|
add_ui_self( R_TapePtr, S_(Binds_SyncPrimitiveArena)),
|
||||||
|
/* Calculate byte offset and store directly back to RAM */
|
||||||
|
sub_u( R_T0, R_PrimCursor, R_T0), // R_T0 = R_PrimCursor - binds.cursor
|
||||||
|
store_word(R_T0, R_AT, 0), // R_AT[0] = R_T0
|
||||||
|
mac_yield()
|
||||||
|
};
|
||||||
|
|
||||||
|
#pragma endregion Baked Atoms
|
||||||
@@ -0,0 +1,564 @@
|
|||||||
|
#pragma region Vendors
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
// #include <assert.h>
|
||||||
|
// #include "libgpu.h"
|
||||||
|
// #include "libetc.h"
|
||||||
|
// #include "libgte.h"
|
||||||
|
#pragma endregion Vendors
|
||||||
|
|
||||||
|
#pragma region Duffle Headers
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
|
||||||
|
#include "duffle/word_count.metadata.h"
|
||||||
|
|
||||||
|
#include "duffle/dsl.h"
|
||||||
|
#include "duffle/memory.h"
|
||||||
|
#include "duffle/math.h"
|
||||||
|
|
||||||
|
#include "duffle/gcc_asm.h"
|
||||||
|
#include "duffle/mips.h"
|
||||||
|
#include "duffle/gp.h"
|
||||||
|
#include "duffle/gte.h"
|
||||||
|
#include "duffle/pad.h"
|
||||||
|
|
||||||
|
#include "duffle/dsl.atom.h"
|
||||||
|
#include "duffle/lottes_tape.h"
|
||||||
|
|
||||||
|
#include "duffle/bios.h"
|
||||||
|
#include "duffle/psyq.h"
|
||||||
|
#pragma endregion Duffle Headers
|
||||||
|
|
||||||
|
#pragma region Duffle TUs
|
||||||
|
#include "duffle/pad.c"
|
||||||
|
#include "duffle/math.atom.c"
|
||||||
|
#include "duffle/mips.atom.c"
|
||||||
|
#include "duffle/gte.atom.c"
|
||||||
|
#include "duffle/gp.atom.c"
|
||||||
|
#include "duffle/pad.atom.c"
|
||||||
|
#include "duffle/psyq.atom.c"
|
||||||
|
#pragma endregion Duffle TUs
|
||||||
|
|
||||||
|
#pragma region Hello Camera Headers
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gen/auto_reg.h"
|
||||||
|
|
||||||
|
#include "hello_camera.h"
|
||||||
|
#pragma endregion Hello Camera Headers
|
||||||
|
|
||||||
|
#pragma region Hello Joypad TUs
|
||||||
|
#include "hello_camera.atom.c"
|
||||||
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
|
enum {
|
||||||
|
Scratchpad_Loc = 0x1F800000,
|
||||||
|
};
|
||||||
|
#define C_scratch(type) C_(type, Scratchpad_Loc)
|
||||||
|
|
||||||
|
enum {
|
||||||
|
Scratchpad_Len = 1024,
|
||||||
|
MemTape_Len = 512,
|
||||||
|
ResolveLookAtArena_Words = 1024,
|
||||||
|
ResolveLookAtArena_Size = ResolveLookAtArena_Words * S_(MipsCode),
|
||||||
|
};
|
||||||
|
typedef Struct_(SMemory) {
|
||||||
|
PrimitiveArena primitives;
|
||||||
|
A2_OrderingTable_Buffer ordering_tbl;
|
||||||
|
DoubleBuffer screen_buf;
|
||||||
|
S4 active_buf_id;
|
||||||
|
|
||||||
|
U4 MemTape[MemTape_Len];
|
||||||
|
|
||||||
|
MT3_S2S4 tform_world;
|
||||||
|
MT3_S2S4 tform_view;
|
||||||
|
|
||||||
|
Camera cam;
|
||||||
|
|
||||||
|
Ent_Cube cube;
|
||||||
|
Ent_Floor floor;
|
||||||
|
|
||||||
|
PadBiosRaw pad_raw[2];
|
||||||
|
PadState pad[2];
|
||||||
|
|
||||||
|
// TODO(Ed): We don't need this we can just cast at any point an address to a desired view of scratchpad, we have the address.
|
||||||
|
U4_V scratchpad; // d-cache
|
||||||
|
|
||||||
|
U1 resolve_look_at_mem[ResolveLookAtArena_Size];
|
||||||
|
MipsAtom* resolve_look_at_atom_addrs[10];
|
||||||
|
};
|
||||||
|
global SMemory smem;
|
||||||
|
extern SMemory smem;
|
||||||
|
|
||||||
|
#define pad0_btn_(btn) btn & smem.pad[0].buttons
|
||||||
|
#define pad1_btn_(btn) btn & smem.pad[1].buttons
|
||||||
|
|
||||||
|
I_ B1* prim__alloc(U4 type_width, Str8 type_name) {
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
gknown B1* buf = (B1*) r_(smem.primitives.buf)[smem.active_buf_id];
|
||||||
|
assert(pa->used + type_width < PrimitiveBuff_Len);
|
||||||
|
B1* next = buf + pa->used;
|
||||||
|
pa->used += type_width;
|
||||||
|
return next;
|
||||||
|
}
|
||||||
|
#define prim_alloc(type) (type*)prim__alloc(S_(type), slit( stringify(type)))
|
||||||
|
|
||||||
|
I_ void resolve_look_at_c11(MT3_S2S4* look_at, P3_S4* eye, P3_S4* target, V3_S4* up_in) {
|
||||||
|
// RGA(Lengyel): Build matrix expansion of a rigid transformation. Corresponding motor is not constructed; we write the LA form for GTE.
|
||||||
|
// Preconditions: eye != target, up_in not collinear with (target - eye).
|
||||||
|
V3_S4 right, up, forward;
|
||||||
|
V3_S4 ux, uy, uz;
|
||||||
|
V3_S4 pos, off;
|
||||||
|
|
||||||
|
forward = target[0]; sub_v3s4(& forward, eye[0]); // RGA(Lengyel): Affine point - point = zero-weight direction.
|
||||||
|
normalize_v3s4(& forward, & uz); // RGA(Lengyel): Normalize the direction bulk. Not finite-point unitization.
|
||||||
|
|
||||||
|
cross_v3s4(& uz, up_in, & right); normalize_v3s4(& right, & ux); // RGA(Lengyel): Complement(Wedge(forward, up_in)) -> right axis.
|
||||||
|
cross_v3s4(& uz, & ux, & up); normalize_v3s4(& up, & uy); // RGA(Lengyel): Complement(Wedge(forward, right)) -> up axis.
|
||||||
|
|
||||||
|
// RGA(Lengyel): matrix expansion of the world-to-camera rotation (basis rows).
|
||||||
|
look_at->m[0][0] = ux.x; look_at->m[0][1] = ux.y; look_at->m[0][2] = ux.z;
|
||||||
|
look_at->m[1][0] = uy.x; look_at->m[1][1] = uy.y; look_at->m[1][2] = uy.z;
|
||||||
|
look_at->m[2][0] = uz.x; look_at->m[2][1] = uz.y; look_at->m[2][2] = uz.z;
|
||||||
|
|
||||||
|
pos = eye[0]; mul_v3s4(& pos, v3s4(-1,-1,-1)); // RGA(Lengyel): -eye in world coordinates (spatial bulk only; implicit weight is dropped).
|
||||||
|
|
||||||
|
// RGA(Lengyel): R * (-eye) is the full matrix translation column.
|
||||||
|
// Motor translator would store half this displacement in m.xyz; GTE consumes full column.
|
||||||
|
mul_m3s2_v3s4(look_at, & pos, & off);
|
||||||
|
trans_m3s2( look_at, & off);
|
||||||
|
}
|
||||||
|
FI_ void camera_look_at_c11(Camera* c, P3_S4* target, V3_S4* up_in) { resolve_look_at_c11(& c->look_at, & c->pos, target, up_in); }
|
||||||
|
|
||||||
|
/* Pre-build all 7 chain atoms of the resolve_look_at bundle into the static arena.
|
||||||
|
* 4 unique procs in hello_camera.atom.c (chain atoms 0, 2, 4, 6); atoms 1, 3, 5
|
||||||
|
* share the GENERIC normalize_v3s4_proc from gte.atom.c
|
||||||
|
* 0: resolve_look_at__input_and_sub_proc
|
||||||
|
* 1: normalize_v3s4_proc (fwd → uz; offsets 0, 16)
|
||||||
|
* 2: resolve_look_at__cross_uz_up_in_to_right_proc
|
||||||
|
* 3: normalize_v3s4_proc (right → ux; offsets 32, 48)
|
||||||
|
* 4: resolve_look_at__cross_uz_ux_to_up_proc
|
||||||
|
* 5: normalize_v3s4_proc (up → uy; offsets 64, 80)
|
||||||
|
* 6: resolve_look_at__populate_and_translate_proc
|
||||||
|
*/
|
||||||
|
internal void resolve_look_at_init(void) {
|
||||||
|
/* Wrap the static arena in a MipsAtomBuilder. */
|
||||||
|
AtomArena ab = atomarena_make(slice_ut_arr(smem.resolve_look_at_mem));
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.resolve_look_at_atom_addrs));
|
||||||
|
|
||||||
|
U4 pin_mask = regfile_abi_mask | (1 << R_ResolveScratch);
|
||||||
|
RegFile rf = regfile(pin_mask);
|
||||||
|
|
||||||
|
// defer(regfile_reset_mask(& rf, pin_mask)) {
|
||||||
|
// U4 r_target_ptr = regfile_alloc(& rf);
|
||||||
|
// U4 r_eye_ptr = regfile_alloc(& rf);
|
||||||
|
// U4 r_up_in_ptr = regfile_alloc(& rf);
|
||||||
|
// U4 r_tmp0 = regfile_alloc(& rf);
|
||||||
|
// U4 r_tmp1 = regfile_alloc(& rf);
|
||||||
|
// U4 r_tmp2 = regfile_alloc(& rf);
|
||||||
|
// U4 r_tmp3 = regfile_alloc(& rf);
|
||||||
|
// tb_emit_(resolve_look_at__input_and_sub_proc(& ab,
|
||||||
|
// R_ResolveScratch,
|
||||||
|
// r_target_ptr, r_eye_ptr, r_up_in_ptr,
|
||||||
|
// r_tmp0, r_tmp1, r_tmp2, r_tmp3));
|
||||||
|
// }
|
||||||
|
U4 r_target_ptr = regfile_alloc(& rf);
|
||||||
|
U4 r_eye_ptr = regfile_alloc(& rf);
|
||||||
|
U4 r_up_in_ptr = regfile_alloc(& rf);
|
||||||
|
U4 r_tmp0 = regfile_alloc(& rf);
|
||||||
|
U4 r_tmp1 = regfile_alloc(& rf);
|
||||||
|
U4 r_tmp2 = regfile_alloc(& rf);
|
||||||
|
U4 r_tmp3 = regfile_alloc(& rf);
|
||||||
|
smem.resolve_look_at_atom_addrs[0] = resolve_look_at__input_and_sub_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_target_ptr, r_eye_ptr, r_up_in_ptr,
|
||||||
|
r_tmp0, r_tmp1, r_tmp2, r_tmp3);
|
||||||
|
|
||||||
|
/* === ATOM 1: normalize fwd→uz === */
|
||||||
|
U4 r_src_offset = O_(ResolveLookAtScratch, fwd);
|
||||||
|
U4 r_dst_offset = O_(ResolveLookAtScratch, uz);
|
||||||
|
U4 r_src_ptr = R_T0;
|
||||||
|
U4 r_dst_ptr = R_T1;
|
||||||
|
U4 r_tmp = R_T2;
|
||||||
|
U4 r_mac1 = R_T3;
|
||||||
|
U4 r_mac2 = R_T5;
|
||||||
|
U4 r_recip = R_T6;
|
||||||
|
U4 r_norm = R_T7;
|
||||||
|
U4 r_shift = R_V0;
|
||||||
|
U4 r_branch = R_V1;
|
||||||
|
// tb_emit_(
|
||||||
|
smem.resolve_look_at_atom_addrs[1] = normalize_v3s4_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_src_offset, r_dst_offset,
|
||||||
|
r_src_ptr, r_dst_ptr, r_tmp,
|
||||||
|
r_mac1, r_mac2, r_recip, r_norm,
|
||||||
|
r_shift, r_branch);
|
||||||
|
// );
|
||||||
|
|
||||||
|
/* === ATOM 2: cross uz×up_in→right === */
|
||||||
|
U4 r_a_2 = R_T0;
|
||||||
|
U4 r_b_2 = R_T1;
|
||||||
|
U4 r_c_2 = R_T2;
|
||||||
|
U4 r_d_2 = R_T3;
|
||||||
|
U4 r_f_2 = R_T5; /* out ptr (HARDCODED in body: scratch+32) */
|
||||||
|
U4 r_g_2 = R_T6; /* a ptr = scratch+16 */
|
||||||
|
U4 r_h_2 = R_T7; /* b ptr = scratch+128 */
|
||||||
|
smem.resolve_look_at_atom_addrs[2] = resolve_look_at__cross_uz_up_in_to_right_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_a_2, r_b_2, r_c_2, r_d_2, r_f_2, r_g_2, r_h_2);
|
||||||
|
|
||||||
|
/* === ATOM 3: normalize right→ux === */
|
||||||
|
U4 r_src_offset_3 = O_(ResolveLookAtScratch, right);
|
||||||
|
U4 r_dst_offset_3 = O_(ResolveLookAtScratch, ux);
|
||||||
|
U4 r_src_ptr_3 = R_T0;
|
||||||
|
U4 r_dst_ptr_3 = R_T1;
|
||||||
|
U4 r_tmp_3 = R_T2;
|
||||||
|
U4 r_mac1_3 = R_T3;
|
||||||
|
U4 r_mac2_3 = R_T5;
|
||||||
|
U4 r_recip_3 = R_T6;
|
||||||
|
U4 r_norm_3 = R_T7;
|
||||||
|
U4 r_shift_3 = R_V0;
|
||||||
|
U4 r_branch_3 = R_V1;
|
||||||
|
smem.resolve_look_at_atom_addrs[3] = normalize_v3s4_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_src_offset_3, r_dst_offset_3,
|
||||||
|
r_src_ptr_3, r_dst_ptr_3, r_tmp_3,
|
||||||
|
r_mac1_3, r_mac2_3, r_recip_3, r_norm_3,
|
||||||
|
r_shift_3, r_branch_3);
|
||||||
|
|
||||||
|
/* === ATOM 4: cross uz×ux→up === */
|
||||||
|
U4 r_a_4 = R_T0;
|
||||||
|
U4 r_b_4 = R_T1;
|
||||||
|
U4 r_c_4 = R_T2;
|
||||||
|
U4 r_d_4 = R_T3;
|
||||||
|
U4 r_f_4 = R_T5; /* out ptr (HARDCODED: scratch+64) */
|
||||||
|
U4 r_g_4 = R_T6; /* a ptr = scratch+16 */
|
||||||
|
U4 r_h_4 = R_T7; /* b ptr = scratch+48 */
|
||||||
|
smem.resolve_look_at_atom_addrs[4] = resolve_look_at__cross_uz_ux_to_up_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_a_4, r_b_4, r_c_4, r_d_4, r_f_4, r_g_4, r_h_4);
|
||||||
|
|
||||||
|
/* === ATOM 5: normalize up→uy === */
|
||||||
|
U4 r_src_offset_5 = O_(ResolveLookAtScratch, up);
|
||||||
|
U4 r_dst_offset_5 = O_(ResolveLookAtScratch, uy);
|
||||||
|
U4 r_src_ptr_5 = R_T0;
|
||||||
|
U4 r_dst_ptr_5 = R_T1;
|
||||||
|
U4 r_tmp_5 = R_T2;
|
||||||
|
U4 r_mac1_5 = R_T3;
|
||||||
|
U4 r_mac2_5 = R_T5;
|
||||||
|
U4 r_recip_5 = R_T6;
|
||||||
|
U4 r_norm_5 = R_T7;
|
||||||
|
U4 r_shift_5 = R_V0;
|
||||||
|
U4 r_branch_5 = R_V1;
|
||||||
|
smem.resolve_look_at_atom_addrs[5] = normalize_v3s4_proc(& ab,
|
||||||
|
R_ResolveScratch,
|
||||||
|
r_src_offset_5, r_dst_offset_5,
|
||||||
|
r_src_ptr_5, r_dst_ptr_5, r_tmp_5,
|
||||||
|
r_mac1_5, r_mac2_5, r_recip_5, r_norm_5,
|
||||||
|
r_shift_5, r_branch_5);
|
||||||
|
|
||||||
|
/* === ATOM 6a: populate (m[][] from ux/uy/uz, t[]=0) === */
|
||||||
|
U4 r_look_at_6a = R_T0; /* tape pop → look_at* */
|
||||||
|
U4 r_scratch_6a = R_ResolveScratch;
|
||||||
|
U4 r_pux_6a = R_T1;
|
||||||
|
U4 r_puy_6a = R_T3;
|
||||||
|
U4 r_puz_6a = R_T5;
|
||||||
|
U4 r_tmp0_6a = R_T2;
|
||||||
|
U4 r_tmp1_6a = R_T6;
|
||||||
|
U4 r_tmp2_6a = R_V0;
|
||||||
|
smem.resolve_look_at_atom_addrs[6] = resolve_look_at__populate_proc(& ab,
|
||||||
|
r_look_at_6a, r_scratch_6a,
|
||||||
|
r_pux_6a, r_puy_6a, r_puz_6a,
|
||||||
|
r_tmp0_6a, r_tmp1_6a, r_tmp2_6a);
|
||||||
|
|
||||||
|
/* === ATOM 6a.5: set_gte_mt3s2s4 (BAKED — ctc2 RT matrix) ===
|
||||||
|
* This is a BAKED atom from gte.atom.c. Its body hardcodes R_T3 as
|
||||||
|
* the matrix pointer (popped from tape). It does NOT need GPR
|
||||||
|
* assignment from us — it has its own internal GPR usage.
|
||||||
|
* We just take its address. */
|
||||||
|
smem.resolve_look_at_atom_addrs[7] = (MipsAtom*) & set_gte_mt3s2s4;
|
||||||
|
|
||||||
|
/* === ATOM 6b: matrix_vector (RT * (-eye) >> 12) ===
|
||||||
|
* Uses mac_apply_matrix_lv component macro which internally uses
|
||||||
|
* r_t0 for the RT matrix load + V0 load, then r_t0/r_t1/r_t2
|
||||||
|
* for the mfc2/store. We pass our GPRs. */
|
||||||
|
U4 r_scratch_6b = R_ResolveScratch;
|
||||||
|
U4 r_peye_6b = R_T1; /* scratch+96 (packed V0 dst, then off dst) */
|
||||||
|
U4 r_look_at_6b = R_T0; /* tape pop → look_at* */
|
||||||
|
U4 r_tmp0_6b = R_T2;
|
||||||
|
U4 r_tmp1_6b = R_T3;
|
||||||
|
U4 r_tmp2_6b = R_T5;
|
||||||
|
smem.resolve_look_at_atom_addrs[8] = resolve_look_at__matrix_vector_proc(& ab,
|
||||||
|
r_scratch_6b, r_peye_6b, r_look_at_6b,
|
||||||
|
r_tmp0_6b, r_tmp1_6b, r_tmp2_6b);
|
||||||
|
|
||||||
|
/* === ATOM 6c: trans_matrix (off → look_at->t[]) === */
|
||||||
|
U4 r_look_at_6c = R_T0; /* tape pop → look_at* */
|
||||||
|
U4 r_scratch_6c = R_ResolveScratch;
|
||||||
|
U4 r_off_ptr_6c = R_T1; /* &scratch.eye (= off dst) */
|
||||||
|
U4 r_tmp0_6c = R_T2;
|
||||||
|
smem.resolve_look_at_atom_addrs[9] = resolve_look_at__trans_matrix_proc(& ab,
|
||||||
|
r_look_at_6c, r_scratch_6c, r_off_ptr_6c, r_tmp0_6c, R_T3, R_T4);
|
||||||
|
|
||||||
|
/* Sanity check: arena didn't overflow. */
|
||||||
|
assert(ab.used <= ResolveLookAtArena_Size);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Emit the resolve_look_at bundle into the tape. Called once per frame from update().
|
||||||
|
* The 7 chain atoms are pre-built at init time (resolve_look_at_init) and referenced by address via smem.resolve_look_at_atom_addrs[].
|
||||||
|
* Per-frame work: 7 tb_emit (atom pointer emissions) + 5 tb_data (C-side pointers for atom 0 + look_at for atom 6).
|
||||||
|
*
|
||||||
|
* Binds_ contract (the field-name labels are for human readability):
|
||||||
|
* Atom 0 input_and_sub target(4) eye(4) up_in(4) scratch_base(4) = 4 words
|
||||||
|
* Atoms 1-5 (no tape data — atom uses r_scratch + offset internally)
|
||||||
|
* Atom 6 populate_and_translate look_at(4) = 1 word
|
||||||
|
* ----
|
||||||
|
* 5 tb_data words total per frame.
|
||||||
|
*/
|
||||||
|
I_ void resolve_look_at(
|
||||||
|
TapeBuilder_R tb
|
||||||
|
, MT3_S2S4* look_at
|
||||||
|
, P3_S4* eye
|
||||||
|
, P3_S4* target
|
||||||
|
, V3_S4* up_in
|
||||||
|
){
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[0]); {
|
||||||
|
tb_data(tb, u4_(target));
|
||||||
|
tb_data(tb, u4_(eye));
|
||||||
|
tb_data(tb, u4_(up_in));
|
||||||
|
tb_data(tb, u4_(smem.scratchpad));
|
||||||
|
}
|
||||||
|
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[1]); { }
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[2]); { }
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[3]); { }
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[4]); { }
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[5]); { }
|
||||||
|
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[6]); {
|
||||||
|
tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[7]); {
|
||||||
|
tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[8]); {
|
||||||
|
tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
tb_emit(tb, smem.resolve_look_at_atom_addrs[9]); {
|
||||||
|
// tb_data(tb, u4_(look_at));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
GCC_OPTIMIZATION_DISABLE
|
||||||
|
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||||
|
{
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape));
|
||||||
|
|
||||||
|
// Pad Input
|
||||||
|
{
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
// Grab latest state from bios.
|
||||||
|
tb_emit_(pad_bios_snapshot);
|
||||||
|
tb_data_(raw, & smem.pad_raw[0]);
|
||||||
|
tb_data_(state, & smem.pad[0]);
|
||||||
|
// tb_emit_(pad_bios_snapshot);
|
||||||
|
// tb_data_(raw, & smem.pad_raw[1]);
|
||||||
|
// tb_data_(state, & smem.pad[1]);
|
||||||
|
|
||||||
|
tb_emit_(pad_input_cam);
|
||||||
|
tb_data_(state, & smem.pad[0]);
|
||||||
|
tb_data_(cam, & smem.cam);
|
||||||
|
|
||||||
|
// tb_emit_(pad_input_cube_rotation);
|
||||||
|
// tb_data_(state, & smem.pad[0]);
|
||||||
|
// tb_data_(cube_rot, & smem.cube.rot);
|
||||||
|
// tb_data_(floor_rot, & smem.floor.rot);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
orderingtbl_clear_reverse(ordering_buf, OrderingTbl_Len);
|
||||||
|
|
||||||
|
// Update the position based on acceleration and velocity
|
||||||
|
gknown V3_S4_R pos = & smem.cube.pos;
|
||||||
|
gknown V3_S4_R vel = & smem.cube.vel;
|
||||||
|
gknown V3_S4_R acc = & smem.cube.accel;
|
||||||
|
add_v3s4(vel, acc[0]);
|
||||||
|
add_v3s4_fp(pos, vel[0]);
|
||||||
|
// vel->x += acc->x;
|
||||||
|
// vel->y += acc->y;
|
||||||
|
// vel->z += acc->z;
|
||||||
|
// pos->x += vel->x;
|
||||||
|
// pos->y += vel->y;
|
||||||
|
// pos->z += vel->z;
|
||||||
|
|
||||||
|
if (pos->y + 150 > smem.floor.pos.y) vel->y *= -1;
|
||||||
|
|
||||||
|
// Prep
|
||||||
|
S4 nclip = 0;
|
||||||
|
S4 orderingtbl_z = 0;
|
||||||
|
A2_S2 p; //???
|
||||||
|
S4 flag; //????
|
||||||
|
|
||||||
|
B4 use_c11_path = false;
|
||||||
|
if (use_c11_path) {
|
||||||
|
camera_look_at_c11(& smem.cam, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||||
|
}
|
||||||
|
if (use_c11_path == false)
|
||||||
|
{
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
resolve_look_at(& tb, & smem.cam.look_at, & smem.cam.pos, & smem.cube.pos, & v3s4(0, -fp_one, 0));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Draw cube
|
||||||
|
if (1)
|
||||||
|
{
|
||||||
|
mt3s2s4_rotation (& smem.cube.rot, & smem.tform_world);
|
||||||
|
mt3s2s4_translation(& smem.tform_world, & smem.cube.pos);
|
||||||
|
mt3s2s4_scale (& smem.tform_world, & smem.cube.scale);
|
||||||
|
|
||||||
|
// Combine world and look_at matrix.
|
||||||
|
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||||
|
gte_matrix_set_rotation (& smem.tform_view);
|
||||||
|
gte_matrix_set_translation(& smem.tform_view);
|
||||||
|
|
||||||
|
// gte_matrix_set_rotation (& smem.tform_world);
|
||||||
|
// gte_matrix_set_translation(& smem.tform_world);
|
||||||
|
|
||||||
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
|
tb.used = 0; tb_scope(& tb) {
|
||||||
|
tb_emit(& tb, rbind_cube_g4_face);
|
||||||
|
tb_data(& tb, prim_cursor);
|
||||||
|
tb_data(& tb, u4_(smem.cube.faces));
|
||||||
|
tb_data(& tb, u4_(smem.cube.verts));
|
||||||
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
|
|
||||||
|
for (U4 i = 0; i < Cube_num_faces; i++) {
|
||||||
|
// Two triangles per quad face: (x,y,z) and (x,z,w)
|
||||||
|
tb_emit(& tb, cube_g4_face);
|
||||||
|
}
|
||||||
|
|
||||||
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
|
tb_data(& tb, u4_(& pa->used));
|
||||||
|
tb_data(& tb, prim_base);
|
||||||
|
}
|
||||||
|
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||||
|
|
||||||
|
// smem.cube.rot.y += 30;
|
||||||
|
}
|
||||||
|
// Draw floor
|
||||||
|
if (1)
|
||||||
|
{
|
||||||
|
mt3s2s4_rotation (& smem.floor.rot, & smem.tform_world);
|
||||||
|
mt3s2s4_translation(& smem.tform_world, & smem.floor.pos);
|
||||||
|
mt3s2s4_scale (& smem.tform_world, & smem.floor.scale);
|
||||||
|
|
||||||
|
// Combine world and look_at matrix.
|
||||||
|
gte_comp_coord_m3s2(& smem.cam.look_at, & smem.tform_world, & smem.tform_view);
|
||||||
|
|
||||||
|
gte_matrix_set_rotation (& smem.tform_view);
|
||||||
|
gte_matrix_set_translation(& smem.tform_view);
|
||||||
|
|
||||||
|
U4 prim_base = u4_(pa->buf[smem.active_buf_id]);
|
||||||
|
U4 prim_cursor = prim_base + pa->used;
|
||||||
|
|
||||||
|
// TODO(Ed): We should do a bounds check beforehand to confirm pa can hold all tris?
|
||||||
|
// The tape atoms in-flight should not need to care.
|
||||||
|
|
||||||
|
// Prepare the tape. (Push protocol to tape)
|
||||||
|
tb.used = 0; tb_scope(& tb) {
|
||||||
|
// tb_emit(& tb, set_gte_mt3s2s4);
|
||||||
|
// tb_data(& tb, u4_(& smem.tform_view));
|
||||||
|
|
||||||
|
tb_emit(& tb, rbind_floor_f3_face);
|
||||||
|
// TODO(Ed): Just use a single context struct ref?
|
||||||
|
tb_data(& tb, prim_cursor);
|
||||||
|
tb_data(& tb, u4_(smem.floor.faces));
|
||||||
|
tb_data(& tb, u4_(smem.floor.verts));
|
||||||
|
tb_data(& tb, u4_(ordering_buf));
|
||||||
|
for (U4 i = 0; i < Floor_num_faces; i++) {
|
||||||
|
tb_emit(& tb, floor_f3_face);
|
||||||
|
}
|
||||||
|
// After floor_f3_face iterations complete, the primitive arena's used counter needs updating.
|
||||||
|
tb_emit(& tb, sync_primitive_arena);
|
||||||
|
tb_data(& tb, u4_(& pa->used));
|
||||||
|
tb_data(& tb, prim_base);
|
||||||
|
}
|
||||||
|
tape_run_a02_s07(tb_slice(tb));// Fire off the tape (bigger-clobber variant).
|
||||||
|
|
||||||
|
// C-side state (pa->used) has already been updated by the tape!
|
||||||
|
// smem.floor.rot.y += 5;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|
||||||
|
void render(void) {
|
||||||
|
}
|
||||||
|
|
||||||
|
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||||
|
draw_sync(0);
|
||||||
|
vsync(0);
|
||||||
|
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||||
|
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||||
|
{
|
||||||
|
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||||
|
pa->used = 0;
|
||||||
|
}
|
||||||
|
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||||
|
}
|
||||||
|
|
||||||
|
GCC_OPTIMIZATION_DISABLE
|
||||||
|
int main(void)
|
||||||
|
{
|
||||||
|
smem = (SMemory){0};
|
||||||
|
// TODO(Ed): remove this field we don't need it in smem.
|
||||||
|
smem.scratchpad = C_(U4_V, Scratchpad_Loc);
|
||||||
|
// smem.primitives.used = 0;
|
||||||
|
// smem.active_buf_id = 0;
|
||||||
|
smem.cam.pos = v3s4(500, -1000, -1500);
|
||||||
|
/*Persistent Entity Setup*/{
|
||||||
|
ent_cube128_init(& smem.cube.verts, & smem.cube.faces); {
|
||||||
|
Ent_Cube* cube = & smem.cube;
|
||||||
|
cube->rot = v3s2(0, 0, 0);
|
||||||
|
cube->scale = v3s4_fp_one();
|
||||||
|
cube->accel = v3s4(0, 1, 0);
|
||||||
|
cube->pos = v3s4(0, -400, 1800);
|
||||||
|
}
|
||||||
|
ent_floor_init(& smem.floor.verts, & smem.floor.faces); {
|
||||||
|
Ent_Floor* floor = & smem.floor;
|
||||||
|
floor->rot = v3s2(0, 0, 0);
|
||||||
|
floor->pos = v3s4(0, 450, 1800);
|
||||||
|
floor->scale = v3s4_fp_one();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
TapeBuilder tb = tb_make(slice_ut_arr(smem.MemTape)); {
|
||||||
|
reset_graph(0);
|
||||||
|
/* Direct BIOS: poll both ports during VBlank. */
|
||||||
|
pad_bios_init_start(& smem.pad_raw[0], & smem.pad_raw[1]);
|
||||||
|
|
||||||
|
/* Pre-build the resolve_look_at bundle atoms into the static arena. */
|
||||||
|
resolve_look_at_init();
|
||||||
|
|
||||||
|
/* Pinned registers for the GPU init atom. */
|
||||||
|
register U4* io_base_addr rgcc(R_IO_BaseAddr) = u4_r(IO_BASE_ADDR);
|
||||||
|
register DoubleBuffer* screen_buf rgcc(R_ScreenBuf) = & smem.screen_buf;
|
||||||
|
tb.used = 0; tb_scope_run(& tb) {
|
||||||
|
tb_emit(& tb, screen_env_init);
|
||||||
|
tb_emit(& tb, gp_screen_init);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
while (1) {
|
||||||
|
gknown S4* active_buf_id = & smem.active_buf_id;
|
||||||
|
gknown U4* ordering_buf = r_(smem.ordering_tbl)[active_buf_id[0]];
|
||||||
|
gknown PrimitiveArena* pa = & smem.primitives;
|
||||||
|
update(pa, ordering_buf);
|
||||||
|
render();
|
||||||
|
gp_display_frame(& smem.screen_buf, active_buf_id, ordering_buf, pa);
|
||||||
|
};
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
# pragma once
|
||||||
|
# include "duffle/dsl.h"
|
||||||
|
# include "duffle/math.h"
|
||||||
|
# include "duffle/gp.h"
|
||||||
|
# include "duffle/pad.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
enum {
|
||||||
|
// PrimitiveBuff_Len = 4096,
|
||||||
|
// OrderingTbl_Len = 2048,
|
||||||
|
PrimitiveBuff_Len = 131072,
|
||||||
|
OrderingTbl_Len = 8192,
|
||||||
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
ScreenRes_X = 320,
|
||||||
|
ScreenRes_Y = 240,
|
||||||
|
ScreenZ = 320,
|
||||||
|
ScreenRes_CenterX = (ScreenRes_X >> 1),
|
||||||
|
ScreenRes_CenterY = (ScreenRes_Y >> 1),
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
|
|
||||||
|
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||||
|
typedef Array_(PrimitiveBuffer, 2);
|
||||||
|
typedef Struct_(PrimitiveArena) {
|
||||||
|
A2_PrimitiveBuffer buf;
|
||||||
|
U4 used;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Cube_num_verts 8
|
||||||
|
typedef Array_(V3_S2, Cube_num_verts);
|
||||||
|
#define Cube_num_faces 6
|
||||||
|
typedef Array_(V4_S2, Cube_num_faces);
|
||||||
|
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||||
|
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||||
|
{ -128, -128, -128 },
|
||||||
|
{ 128, -128, -128 },
|
||||||
|
{ 128, -128, 128 },
|
||||||
|
{ -128, -128, 128 },
|
||||||
|
{ -128, 128, -128 },
|
||||||
|
{ 128, 128, -128 },
|
||||||
|
{ 128, 128, 128 },
|
||||||
|
{ -128, 128, 128 }
|
||||||
|
};
|
||||||
|
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||||
|
{ 3, 2, 0, 1 },
|
||||||
|
{ 0, 1, 4, 5 },
|
||||||
|
{ 4, 5, 7, 6 },
|
||||||
|
{ 1, 2, 5, 6 },
|
||||||
|
{ 2, 3, 6, 7 },
|
||||||
|
{ 3, 0, 7, 4 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
typedef Struct_(Ent_Cube) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 vel;
|
||||||
|
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A8_V3_S2 verts;
|
||||||
|
A6_V4_S2 faces;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Floor_num_verts 4
|
||||||
|
typedef Array_(V3_S2, Floor_num_verts);
|
||||||
|
#define Floor_num_faces 2
|
||||||
|
typedef Array_(V3_S2, Floor_num_faces);
|
||||||
|
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||||
|
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||||
|
{ -900, 0, -900 },
|
||||||
|
{ -900, 0, 900 },
|
||||||
|
{ 900, 0, -900 },
|
||||||
|
{ 900, 0, 900 },
|
||||||
|
};
|
||||||
|
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||||
|
{ 0, 1, 2 },
|
||||||
|
{ 1, 3, 2 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||||
|
};
|
||||||
|
typedef Struct_(Ent_Floor) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A4_V3_S2 verts;
|
||||||
|
A2_V3_S2 faces;
|
||||||
|
};
|
||||||
|
|
||||||
|
typedef Struct_(Camera) {
|
||||||
|
P3_S4 pos; // RGA(Lengyel): affine point with implicit weight one. Storage alias of V3_S4.
|
||||||
|
V3_S2 rot;
|
||||||
|
MT3_S2S4 look_at;
|
||||||
|
};
|
||||||
@@ -14,13 +14,13 @@
|
|||||||
#include "duffle/gp.h"
|
#include "duffle/gp.h"
|
||||||
#include "duffle/gte.h"
|
#include "duffle/gte.h"
|
||||||
|
|
||||||
# include "duffle/gen/duffle.macs.h"
|
# include "duffle/gen/macs.h"
|
||||||
# include "duffle/gen/duffle.offsets.h"
|
# include "duffle/gen/offsets.h"
|
||||||
#include "duffle/atom_dsl.h"
|
#include "duffle/atom_dsl.h"
|
||||||
#include "duffle/lottes_tape.h"
|
#include "duffle/lottes_tape.h"
|
||||||
#include "duffle/word_count.metadata.h"
|
#include "duffle/word_count.metadata.h"
|
||||||
|
|
||||||
# include "gen/hello_gte.offsets.h"
|
# include "gen/offsets.h"
|
||||||
#include "hello_gte.h"
|
#include "hello_gte.h"
|
||||||
|
|
||||||
#include "hello_gte.tape.c"
|
#include "hello_gte.tape.c"
|
||||||
|
|||||||
@@ -1,10 +1,10 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# include "duffle/gen/duffle.macs.h"
|
# include "duffle/gen/macs.h"
|
||||||
# include "duffle/gen/duffle.offsets.h"
|
# include "duffle/gen/offsets.h"
|
||||||
# include "duffle/atom_dsl.h"
|
# include "duffle/atom_dsl.h"
|
||||||
# include "duffle/lottes_tape.h"
|
# include "duffle/lottes_tape.h"
|
||||||
# include "duffle/word_count.metadata.h"
|
# include "duffle/word_count.metadata.h"
|
||||||
# include "gen/hello_gte.offsets.h"
|
# include "gen/offsets.h"
|
||||||
# include "hello_gte.h"
|
# include "hello_gte.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,41 @@
|
|||||||
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
|
#pragma once
|
||||||
|
#endif
|
||||||
|
// Auto-generated by ps1_meta.lua — DO NOT EDIT
|
||||||
|
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad/
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
|
||||||
|
// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)
|
||||||
|
|
||||||
|
#ifndef WORD_COUNT
|
||||||
|
#define WORD_COUNT(name, count) enum { words_##name = (count) };
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#define mac_put_disp_env(reg_transfer, reg_base, port) \
|
||||||
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_set_mask_bit(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_put_disp_env, 5)
|
||||||
|
|
||||||
|
#define mac_put_draw_env(reg_transfer, reg_base, port) \
|
||||||
|
mac_gcmd_push(gp0_dr_env_tag, reg_transfer, reg_base, port) /* tag (length=15 << 24, addr=0) — packet header for the DR_ENV sequence. The GPU needs this to recognize the next 15 words as a DR_ENV packet and trigger the isbg auto-clear. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_mode_drawing_allowed, reg_transfer, reg_base, port) /* code[0] DrawMode (dfe=1, dtd=0, tpage=0) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[1] TextureWindow (tw=(0,0)) */ \
|
||||||
|
, mac_gcmd_push(enc_gp0_draw_area_tl_word(0, ScreenRes_Y), reg_transfer, reg_base, port) /* code[2] DrawArea top-left (clip.x=0, clip.y=ScreenRes_Y=240) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port) /* code[3] DrawArea bottom-right (clip.x+w=320, clip.y+h=480) */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_draw_offset(), reg_transfer, reg_base, port) /* code[4] DrawOffset (ofs=(0,0)) — bare-cmd word; the GPU uses the current state machine. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_mask(), reg_transfer, reg_base, port) /* code[5] Mask (dtd=0, dfe=1, isbg=1) — 0xE6 cmd + isbg bit. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_bg_color_cmd(1, 7, 7, 7), reg_transfer, reg_base, port) /* code[6] Initial-bg-color + auto-clear (isbg=1, r=7, g=7, b=7). */ \
|
||||||
|
, mac_gcmd_push(gp0_word_dr_env_draw_mode(1), reg_transfer, reg_base, port) /* code[7] Re-assert DrawMode with isbg=1 (isbg-flag set; the 0xE1 cmd byte plus isbg only). */ /* code[8..10] Padding (NOP — GPU discards; the DR_ENV requires 16 words total). */ \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) /* code[11..12] TextureWindow bottom-right (tw.x+tw.w=0, tw.y+tw.h=0) — libpsyx emits twice. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_set_texture_window(), reg_transfer, reg_base, port) /* code[13..14] Padding (NOP) — completes the 16-word packet. */ \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port) \
|
||||||
|
, mac_gcmd_push(gp0_word_nop(), reg_transfer, reg_base, port)
|
||||||
|
WORD_COUNT(mac_put_draw_env, 16)
|
||||||
|
|
||||||
@@ -1,8 +1,11 @@
|
|||||||
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT
|
||||||
// Source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.tape.c
|
// Directory: C:\projects\Pikuma\ps1\code\hello_joypad\
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.c
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.h
|
||||||
|
// source: C:\projects\Pikuma\ps1\code\hello_joypad\hello_joypad.atom.c
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#pragma region hello_joypad.tape
|
#pragma region hello_joypad
|
||||||
|
|
||||||
|
|
||||||
// --- atom: cube_g4_face (76 words) ---
|
// --- atom: cube_g4_face (76 words) ---
|
||||||
@@ -25,18 +28,18 @@ enum {
|
|||||||
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
atom_offset_bounds_chk_floor_f3_face_exit = _atom_offset_bounds_chk_floor_f3_face_exit,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- atom: pad_bios_snapshot (77 words) ---
|
// --- atom: pad_bios_snapshot (78 words) ---
|
||||||
|
|
||||||
#define _atom_offset_snap_root_skip_disconnected 8
|
#define _atom_offset_snap_root_skip_disconnected 8
|
||||||
#define _atom_offset_disconnected_snap_end 60
|
#define _atom_offset_disconnected_snap_end 61
|
||||||
#define _atom_offset_case_2_id_dispatch 8
|
#define _atom_offset_case_2_id_dispatch 8
|
||||||
#define _atom_offset_pending_snap_end 50
|
#define _atom_offset_pending_snap_end 51
|
||||||
#define _atom_offset_id_dispatch_try_analog_stick 11
|
#define _atom_offset_id_dispatch_try_analog_stick 11
|
||||||
#define _atom_offset_id_dispatch_snap_end 37
|
#define _atom_offset_id_dispatch_snap_end 38
|
||||||
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
#define _atom_offset_try_analog_stick_try_analog_pad 12
|
||||||
#define _atom_offset_analog_stick_snap_end 23
|
#define _atom_offset_analog_stick_snap_end 24
|
||||||
#define _atom_offset_try_analog_pad_try_unsupported 11
|
#define _atom_offset_try_analog_pad_try_unsupported 11
|
||||||
#define _atom_offset_analog_pad_snap_end 9
|
#define _atom_offset_analog_pad_snap_end 10
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
atom_offset_snap_root_skip_disconnected = _atom_offset_snap_root_skip_disconnected,
|
||||||
@@ -51,14 +54,14 @@ enum {
|
|||||||
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
atom_offset_analog_pad_snap_end = _atom_offset_analog_pad_snap_end,
|
||||||
};
|
};
|
||||||
|
|
||||||
// --- atom: pad_apply_input (59 words) ---
|
// --- atom: pad_apply_input (60 words) ---
|
||||||
|
|
||||||
#define _atom_offset_dpad_left_exit_dpad_left 6
|
#define _atom_offset_dpad_left_exit_dpad_left 6
|
||||||
#define _atom_offset_dpad_right_exit_dpad_right 6
|
#define _atom_offset_dpad_right_exit_dpad_right 6
|
||||||
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
#define _atom_offset_dead_zone_low_check_dead_low_active 8
|
||||||
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
#define _atom_offset_dead_zone_high_check_dead_high_active 15
|
||||||
#define _atom_offset_dead_zone_skip_exit_stick 23
|
#define _atom_offset_dead_zone_skip_exit_stick 24
|
||||||
#define _atom_offset_end_low_exit_stick 11
|
#define _atom_offset_end_low_exit_stick 12
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
atom_offset_dpad_left_exit_dpad_left = _atom_offset_dpad_left_exit_dpad_left,
|
||||||
@@ -69,5 +72,5 @@ enum {
|
|||||||
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
atom_offset_end_low_exit_stick = _atom_offset_end_low_exit_stick,
|
||||||
};
|
};
|
||||||
|
|
||||||
#pragma endregion hello_joypad.tape
|
#pragma endregion hello_joypad
|
||||||
|
|
||||||
@@ -1,53 +1,31 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
#ifdef INTELLISENSE_DIRECTIVES
|
||||||
# include "duffle/gen/duffle.macs.h"
|
# pragma once
|
||||||
# include "duffle/gen/duffle.offsets.h"
|
# include "duffle/gen/macs.h"
|
||||||
# include "duffle/atom_dsl.h"
|
# include "duffle/gen/offsets.h"
|
||||||
|
# include "duffle/dsl.atom.h"
|
||||||
# include "duffle/lottes_tape.h"
|
# include "duffle/lottes_tape.h"
|
||||||
# include "duffle/mips.h"
|
# include "duffle/mips.h"
|
||||||
# include "duffle/gte.h"
|
# include "duffle/gte.h"
|
||||||
# include "duffle/gp.h"
|
# include "duffle/gp.h"
|
||||||
# include "duffle/pad.h"
|
# include "duffle/pad.h"
|
||||||
# include "duffle/word_count.metadata.h"
|
# include "duffle/word_count.metadata.h"
|
||||||
# include "psyq.h"
|
# include "duffle/psyq.h"
|
||||||
# include "gen/hello_joypad.offsets.h"
|
# include "duffle/math.atom.c"
|
||||||
# include "gen/hello_joypad.macs.h"
|
# include "duffle/mips.atom.c"
|
||||||
|
# include "duffle/gte.atom.c"
|
||||||
|
# include "duffle/gp.atom.c"
|
||||||
|
# include "duffle/psyq.atom.c"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
# include "gen/macs.h"
|
||||||
# include "hello_joypad.h"
|
# include "hello_joypad.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
ATOM_FILE_DEBUGGER_LINE_MARKER(hello_joypad_atom_c);
|
||||||
|
|
||||||
#pragma region MACs (Mips Atom components)
|
#pragma region MACs (Mips Atom components)
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_load_v2s2(U4 rs_x, U4 rs_y, U4 r_base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_load_v2s2, {
|
FI_ Slice_MipsCode ac_put_disp_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
load_half( rs_x, r_base, O_(V3_S2,x)),
|
MipsAtomComp_Proc_(ab, {
|
||||||
load_half( rs_y, r_base, O_(V3_S2,y)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_v2s2(U4 rt_x, U4 rt_y, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_v2s2, {
|
|
||||||
store_half(rt_x, base, offset + O_(V2_S2,x)),
|
|
||||||
store_half(rt_y, base, offset + O_(V2_S2,y)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rects2(U4 rt_x, U4 rt_y, U4 rt_width, U4 rt_height, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rects2, {
|
|
||||||
store_half(rt_x, base, offset + O_(Rect_S2,x)),
|
|
||||||
store_half(rt_y, base, offset + O_(Rect_S2,y)),
|
|
||||||
store_half(rt_width, base, offset + O_(Rect_S2,width)),
|
|
||||||
store_half(rt_height, base, offset + O_(Rect_S2,height)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_store_rgb8(U1 rr, U1 rg, U1 rb, U4 base, U4 offset) atom_dbg_skip MipsAtomComp_Proc_(ac_store_rgb8, {
|
|
||||||
store_byte(rr, base, offset + O_(DrawEnv,initial_bg_color.r)),
|
|
||||||
store_byte(rg, base, offset + O_(DrawEnv,initial_bg_color.g)),
|
|
||||||
store_byte(rb, base, offset + O_(DrawEnv,initial_bg_color.b)),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_gcmd_push(U4 cmd, U4 reg_transfer, U4 reg_base, U2 port)
|
|
||||||
MipsAtomComp_Proc_(ac_gcmd_push, {
|
|
||||||
load_upper_i(reg_transfer, cmd >> 16),
|
|
||||||
or_i_self( reg_transfer, cmd & 0xFFFF),
|
|
||||||
store_word( reg_transfer, reg_base, port),
|
|
||||||
})
|
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_disp_env(U4 reg_transfer, U4 reg_base, U2 port)
|
|
||||||
MipsAtomComp_Proc_(ac_put_disp_env, {
|
|
||||||
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
// Emits 5 GP0 commands for buffer 0 (display_area = (0,0,320,240)).
|
||||||
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
// Sequence per libpsyx PutDispEnv: DrawArea TL → DrawArea BR → Mask → DrawArea TL → DrawArea BR
|
||||||
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_top_left_origin, reg_transfer, reg_base, port),
|
||||||
@@ -57,8 +35,8 @@ MipsAtomComp_Proc_(ac_put_disp_env, {
|
|||||||
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
mac_gcmd_push(gp0_word_draw_area_bottom_right_320x240, reg_transfer, reg_base, port),
|
||||||
})
|
})
|
||||||
|
|
||||||
FI_ Slice_MipsCode ac_put_draw_env(U4 reg_transfer, U4 reg_base, U2 port)
|
FI_ Slice_MipsCode ac_put_draw_env(MipsAtomBuilder_R ab, U4 reg_transfer, U4 reg_base, U2 port)
|
||||||
MipsAtomComp_Proc_(ac_put_draw_env, {
|
MipsAtomComp_Proc_(ab, {
|
||||||
/*
|
/*
|
||||||
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
* ORIGIN: each code word corresponds to the EXACT value libpsyx's PutDrawEnv function would compute for the same DrawEnv settings.
|
||||||
* References:
|
* References:
|
||||||
@@ -138,7 +116,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
|||||||
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
store_word(R_0, R_ScreenBuf, O_(DisplayEnv,vinterlace) + OA_(DoubleBuffer,display,1)),
|
||||||
|
|
||||||
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
mac_store_rects2(R_0, R_ScreenY, R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area) + OA_(DoubleBuffer,draw,0)), /* draw[0].clip_area = (0, 240, 320, 240). C11's SetDefDrawEnv writes clip.y = y_arg. */
|
||||||
mac_store_v2s2( R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
mac_store_v2s2(R_0, R_ScreenY, R_ScreenBuf, O_(DrawEnv,drawing_offset[0]) + OA_(DoubleBuffer,draw,0)), /* draw[0].drawing_offset[0] = (0, 240); C11 passes y_arg as ofs. */
|
||||||
|
|
||||||
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
mac_store_v2s2(R_ScreenX, R_ScreenY, R_ScreenBuf, O_(DrawEnv,clip_area.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
@@ -151,7 +129,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
|||||||
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
store_word(R_0, R_ScreenBuf, O_(DrawEnv,texture_window.width) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
/* draw[0].texture_page = 10 (gp0_tpage_default). C11 SetDefDrawEnv at C11_only.elf:0x8001273C writes the same 0x0A. . */
|
||||||
add_ui(R_T0, R_0, gp0_tpage_default),
|
add_ui(R_T0, R_0, gp0_tpage_default),
|
||||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,0)),
|
||||||
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
store_half(R_T0, R_ScreenBuf, O_(DrawEnv,texture_page) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
@@ -166,7 +144,7 @@ internal MipsAtom_(screen_env_init) atom_info(atom_phase(screen_init)
|
|||||||
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
store_byte(R_T0, R_ScreenBuf, O_(DrawEnv,enable_auto_clear) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
/* draw[0].initial_bg_color = (r=7, g=7, b=7). */
|
||||||
add_ui(R_T0, R_0, 7),
|
add_ui(R_T0, R_0, 7),
|
||||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,0)),
|
||||||
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
mac_store_rgb8(R_T0,R_T0,R_T0, R_ScreenBuf, O_(DrawEnv,initial_bg_color) + OA_(DoubleBuffer,draw,1)),
|
||||||
|
|
||||||
@@ -202,6 +180,17 @@ internal MipsAtom_(gp_screen_init) atom_info(atom_phase(screen_init), atom_reads
|
|||||||
mac_yield(),
|
mac_yield(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
enum {
|
||||||
|
R_PrimCursor = R_T7 atom_reg atom_type(U4*), /* VRAM output cursor (primitive buffer) */
|
||||||
|
R_FaceCursor = R_T4 atom_reg atom_type(V4_S2*), /* Cube face-index cursor (V4_S2*); floor context switches to V3_S2* via atom_phase */
|
||||||
|
R_VertBase = R_T5 atom_reg atom_type(V3_S2*), /* Base address of the vertex array */
|
||||||
|
R_OtBase = R_T6 atom_reg atom_type(U4*), /* Base address of the Ordering Table */
|
||||||
|
#define R_PrimCursor_Code R_T7_Code
|
||||||
|
#define R_FaceCursor_Code R_T4_Code
|
||||||
|
#define R_VertBase_Code R_T5_Code
|
||||||
|
#define R_OtBase_Code R_T6_Code
|
||||||
|
};
|
||||||
|
|
||||||
typedef Struct_(Binds_CubeTri) {
|
typedef Struct_(Binds_CubeTri) {
|
||||||
U4 PrimCursor;
|
U4 PrimCursor;
|
||||||
V4_S2* FaceCursor;
|
V4_S2* FaceCursor;
|
||||||
@@ -232,23 +221,23 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
|||||||
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
load_half_u(R_T2, R_FaceCursor, 2 * S_(S2)),
|
||||||
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
load_half_u(R_T3, R_FaceCursor, 3 * S_(S2)),
|
||||||
|
|
||||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // required cpu -> gte delay slot
|
||||||
gte_cmdw_nclip,
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
gte_mv_from_data_r(R_T0, C2_MAC0), nop,
|
||||||
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
branch_le_zero(R_T0, atom_offset(cull, cube_g4_face_exit)),
|
||||||
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
/* BD-slot: write the prim tag (R_0=0; overwrites the legacy tag word in the prim_buffer).
|
||||||
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded — harmless
|
* If branch IS taken (face culled), the body is skipped and this 0-tag is stranded —
|
||||||
* because the OT entry that points to this prim is created later, only on the body path. */
|
* harmless because the OT entry that points to this prim is created later, only on the body path. */
|
||||||
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
store_word(R_0, R_PrimCursor, O_(Poly_G4, tag)),
|
||||||
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
shift_lleft(R_AT, R_T3, v3s2_byteoff), add_u(R_AT, R_AT, R_VertBase),
|
||||||
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
load_word(R_V0, R_AT, O_(V3_S2, x)), load_word(R_V1, R_AT, O_(V3_S2, z)),
|
||||||
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
gte_mv_to_data_r(R_V0, C2_VXY0), gte_mv_to_data_r(R_V1, C2_VZ0),
|
||||||
|
|
||||||
mac_gte_store_g4_p012(),
|
mac_gte_store_g4_p012(R_PrimCursor),
|
||||||
gte_cmdw_rotate_translate_perspective_single,
|
gte_cmdw_rotate_translate_perspective_single,
|
||||||
mac_gte_store_g4_p3(),
|
mac_gte_store_g4_p3(R_PrimCursor),
|
||||||
|
|
||||||
gte_cmdw_avg_sort_z4,
|
gte_cmdw_avg_sort_z4,
|
||||||
gte_mv_from_data_r(R_T1, C2_OTZ),
|
gte_mv_from_data_r(R_T1, C2_OTZ),
|
||||||
@@ -256,8 +245,8 @@ MipsAtom_(cube_g4_face) atom_info(atom_phase(cube_g4),
|
|||||||
set_lt_u( R_AT, R_T1, R_AT),
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
|
|
||||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, cube_g4_face_exit)), nop,
|
||||||
mac_insert_ot_tag_g4(),
|
mac_insert_ot_tag_g4(R_OtBase, R_PrimCursor),
|
||||||
mac_format_g4_color(
|
mac_format_g4_color(R_PrimCursor,
|
||||||
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
/* c0 magenta */ 0xFF, 0x00, 0xFF,
|
||||||
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
/* c1 yellow */ 0xFF, 0xFF, 0x00,
|
||||||
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
/* c2 cyan */ 0x00, 0xFF, 0xFF,
|
||||||
@@ -297,16 +286,16 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
|||||||
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
, atom_reads( R_PrimCursor, R_FaceCursor, R_VertBase, R_OtBase)
|
||||||
, atom_writes(R_PrimCursor, R_FaceCursor)
|
, atom_writes(R_PrimCursor, R_FaceCursor)
|
||||||
) {
|
) {
|
||||||
mac_load_tri_indices( R_T0, R_T1, R_T2),
|
mac_load_tri_indices(R_FaceCursor, R_T0, R_T1, R_T2),
|
||||||
mac_gte_load_tri_verts(R_T0, R_T1, R_T2),
|
mac_gte_load_tri_verts(R_VertBase, R_T0, R_T1, R_T2),
|
||||||
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
nop2, gte_cmdw_rotate_translate_perspective_triple, // 2 nops retire the final cpu -> gte writes before RTPT
|
||||||
gte_cmdw_nclip,
|
gte_cmdw_nclip,
|
||||||
|
|
||||||
/* Culling (Branch forward if Backface) */
|
/* Culling (Branch forward if Backface) */
|
||||||
gte_mv_from_data_r(R_T0, C2_MAC0),
|
gte_mv_from_data_r(R_T0, C2_MAC0),
|
||||||
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
nop, branch_le_zero(R_T0, atom_offset(culling, floor_f3_face_exit)), nop, // required gte -> cpu load-delay slot.
|
||||||
/* Format Primitive */
|
/* Format Primitive */
|
||||||
mac_gte_store_f3(),
|
mac_gte_store_f3(R_PrimCursor),
|
||||||
|
|
||||||
/* Calculate Depth */
|
/* Calculate Depth */
|
||||||
gte_avg_sort_z3,
|
gte_avg_sort_z3,
|
||||||
@@ -315,8 +304,8 @@ MipsAtom_(floor_f3_face) atom_info(atom_phase(floor_f3)
|
|||||||
add_ui( R_AT, R_0, OrderingTbl_Len),
|
add_ui( R_AT, R_0, OrderingTbl_Len),
|
||||||
set_lt_u( R_AT, R_T1, R_AT),
|
set_lt_u( R_AT, R_T1, R_AT),
|
||||||
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
branch_equal(R_AT, R_0, atom_offset(bounds_chk, floor_f3_face_exit)), nop,
|
||||||
mac_format_f3_color(0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
mac_format_f3_color(R_PrimCursor, 0xFF, 0xFF, 0xFF), // RGB-form (R=FF, G=FF, B=FF = white)
|
||||||
mac_insert_ot_tag_f3(), /* Insert into Ordering Table Linked List */
|
mac_insert_ot_tag_f3(R_OtBase, R_PrimCursor), /* Insert into Ordering Table Linked List */
|
||||||
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
add_ui_self(R_PrimCursor, S_(Poly_F3)), /* Advance Prim Cursor (5 words) */
|
||||||
// Note(Ed): No bounds checking, should be checked before atom runs.
|
// Note(Ed): No bounds checking, should be checked before atom runs.
|
||||||
// end: branch(bounds_chk)
|
// end: branch(bounds_chk)
|
||||||
@@ -504,6 +493,9 @@ atom_label(try_unsupported) /* === Case 7: Unsupported — fall through from the
|
|||||||
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
store_byte( R_T4, R_PadState, O_(PadState,id)),
|
||||||
/* Fall through to snap_end. */
|
/* Fall through to snap_end. */
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(snap_end)
|
atom_label(snap_end)
|
||||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the case-exit branch. */
|
||||||
mac_yield_tail(),
|
mac_yield_tail(),
|
||||||
@@ -635,6 +627,9 @@ atom_label(dead_high_active)
|
|||||||
add_u( R_T0, R_T0, R_T4),
|
add_u( R_T0, R_T0, R_T4),
|
||||||
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
store_half( R_T0, R_FloorRot, O_(V3_S2,y)),
|
||||||
|
|
||||||
|
atom_label(no_jump_fallthrough)
|
||||||
|
mac_yield_load(),
|
||||||
|
|
||||||
atom_label(exit_stick)
|
atom_label(exit_stick)
|
||||||
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
/* NOT mac_yield() — R_AtomJmp was already loaded in the BD-slot of the dead-zone/exit branch. */
|
||||||
mac_yield_tail(),
|
mac_yield_tail(),
|
||||||
@@ -1,9 +1,17 @@
|
|||||||
|
#pragma region Vendors
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
// #include "libgpu.h"
|
// #include "libgpu.h"
|
||||||
// #include "libetc.h"
|
// #include "libetc.h"
|
||||||
// #include "libgte.h"
|
// #include "libgte.h"
|
||||||
|
#pragma endregion Vendors
|
||||||
|
|
||||||
|
#pragma region Duffle Headers
|
||||||
|
# include "duffle/gen/macs.h"
|
||||||
|
# include "duffle/gen/offsets.h"
|
||||||
|
|
||||||
|
#include "duffle/word_count.metadata.h"
|
||||||
|
|
||||||
#include "duffle/dsl.h"
|
#include "duffle/dsl.h"
|
||||||
#include "duffle/memory.h"
|
#include "duffle/memory.h"
|
||||||
@@ -15,96 +23,30 @@
|
|||||||
#include "duffle/gte.h"
|
#include "duffle/gte.h"
|
||||||
#include "duffle/pad.h"
|
#include "duffle/pad.h"
|
||||||
|
|
||||||
# include "duffle/gen/duffle.macs.h"
|
#include "duffle/dsl.atom.h"
|
||||||
# include "duffle/gen/duffle.offsets.h"
|
|
||||||
#include "duffle/atom_dsl.h"
|
|
||||||
#include "duffle/lottes_tape.h"
|
#include "duffle/lottes_tape.h"
|
||||||
#include "duffle/word_count.metadata.h"
|
|
||||||
|
|
||||||
#include "psyq.h"
|
#include "duffle/psyq.h"
|
||||||
|
#pragma endregion Duffle Headers
|
||||||
|
|
||||||
|
#pragma region Duffle TUs
|
||||||
|
#include "duffle/math.atom.c"
|
||||||
|
#include "duffle/mips.atom.c"
|
||||||
|
#include "duffle/gte.atom.c"
|
||||||
|
#include "duffle/gp.atom.c"
|
||||||
|
#include "duffle/psyq.atom.c"
|
||||||
|
#pragma endregion Duffle TUs
|
||||||
|
|
||||||
|
#pragma region Joypade Headers
|
||||||
|
# include "gen/macs.h"
|
||||||
|
# include "gen/offsets.h"
|
||||||
|
|
||||||
# include "gen/hello_joypad.macs.h"
|
|
||||||
# include "gen/hello_joypad.offsets.h"
|
|
||||||
#include "hello_joypad.h"
|
#include "hello_joypad.h"
|
||||||
|
#pragma region Joypad Headers
|
||||||
|
|
||||||
#include "psyq.c"
|
#pragma region Hello Joypad TUs
|
||||||
#include "hello_joypad.tape.c"
|
#include "hello_joypad.atom.c"
|
||||||
|
#pragma endregion Hello Joypad TUs
|
||||||
|
|
||||||
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
|
||||||
typedef Array_(OrderingTable_Buffer, 2);
|
|
||||||
|
|
||||||
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
|
||||||
typedef Array_(PrimitiveBuffer, 2);
|
|
||||||
typedef Struct_(PrimitiveArena) {
|
|
||||||
A2_PrimitiveBuffer buf;
|
|
||||||
U4 used;
|
|
||||||
};
|
|
||||||
|
|
||||||
#define Cube_num_verts 8
|
|
||||||
typedef Array_(V3_S2, Cube_num_verts);
|
|
||||||
#define Cube_num_faces 6
|
|
||||||
typedef Array_(V4_S2, Cube_num_faces);
|
|
||||||
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
|
||||||
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
|
||||||
{ -128, -128, -128 },
|
|
||||||
{ 128, -128, -128 },
|
|
||||||
{ 128, -128, 128 },
|
|
||||||
{ -128, -128, 128 },
|
|
||||||
{ -128, 128, -128 },
|
|
||||||
{ 128, 128, -128 },
|
|
||||||
{ 128, 128, 128 },
|
|
||||||
{ -128, 128, 128 }
|
|
||||||
};
|
|
||||||
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
|
||||||
{ 3, 2, 0, 1 },
|
|
||||||
{ 0, 1, 4, 5 },
|
|
||||||
{ 4, 5, 7, 6 },
|
|
||||||
{ 1, 2, 5, 6 },
|
|
||||||
{ 2, 3, 6, 7 },
|
|
||||||
{ 3, 0, 7, 4 },
|
|
||||||
};
|
|
||||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
|
||||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
typedef Struct_(Ent_Cube) {
|
|
||||||
V3_S4 accel;
|
|
||||||
V3_S4 vel;
|
|
||||||
V3_S4 pos;
|
|
||||||
V3_S4 scale;
|
|
||||||
V3_S2 rot;
|
|
||||||
A8_V3_S2 verts;
|
|
||||||
A6_V4_S2 faces;
|
|
||||||
};
|
|
||||||
|
|
||||||
#define Floor_num_verts 4
|
|
||||||
typedef Array_(V3_S2, Floor_num_verts);
|
|
||||||
#define Floor_num_faces 2
|
|
||||||
typedef Array_(V3_S2, Floor_num_faces);
|
|
||||||
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
|
||||||
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
|
||||||
{ -900, 0, -900 },
|
|
||||||
{ -900, 0, 900 },
|
|
||||||
{ 900, 0, -900 },
|
|
||||||
{ 900, 0, 900 },
|
|
||||||
};
|
|
||||||
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
|
||||||
{ 0, 1, 2 },
|
|
||||||
{ 1, 3, 2 },
|
|
||||||
};
|
|
||||||
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
|
||||||
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
|
||||||
};
|
|
||||||
typedef Struct_(Ent_Floor) {
|
|
||||||
V3_S4 accel;
|
|
||||||
V3_S4 pos;
|
|
||||||
V3_S4 scale;
|
|
||||||
V3_S2 rot;
|
|
||||||
A4_V3_S2 verts;
|
|
||||||
A2_V3_S2 faces;
|
|
||||||
};
|
|
||||||
|
|
||||||
|
|
||||||
enum {
|
enum {
|
||||||
Scratchpad_Len = 1024,
|
Scratchpad_Len = 1024,
|
||||||
@@ -212,50 +154,6 @@ NI_ void pad_bios_init_start(PadBiosRaw* raw0, PadBiosRaw* raw1)
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
void gp_screen_init_c11(DoubleBuffer* screen_buf, S4* active_buf_id)
|
|
||||||
{
|
|
||||||
reset_graph(0);
|
|
||||||
|
|
||||||
// Set the current initial buffer
|
|
||||||
active_buf_id[0] = 0;
|
|
||||||
|
|
||||||
// Just setting env data, not interacting with console hw.
|
|
||||||
// First buffer area
|
|
||||||
displayenv_init(& r_(screen_buf->display)[0], 0, 0, ScreenRes_X, ScreenRes_Y);
|
|
||||||
drawenv_init (& r_(screen_buf->draw )[0], 0, ScreenRes_Y, ScreenRes_X, ScreenRes_Y);
|
|
||||||
// Second buffer area
|
|
||||||
displayenv_init(& r_(screen_buf->display)[1], 0, ScreenRes_Y, ScreenRes_X, ScreenRes_Y);
|
|
||||||
drawenv_init (& r_(screen_buf->draw )[1], 0, 0, ScreenRes_X, ScreenRes_Y);
|
|
||||||
// Set the back/drawing buffer
|
|
||||||
screen_buf->draw[0].enable_auto_clear = true;
|
|
||||||
screen_buf->draw[1].enable_auto_clear = true;
|
|
||||||
// Set the background clear color
|
|
||||||
screen_buf->draw[0].initial_bg_color = rgb8( .r = 7, .g = 7, .b = 7 );
|
|
||||||
screen_buf->draw[1].initial_bg_color = rgb8( .r = 7, .g = 7, .b = 7 );
|
|
||||||
// screen_buf->draw[1].initial_bg_color = rgb8( .r = 47, .g = 13, .b = 0 );
|
|
||||||
displayenv_put(& r_(screen_buf->display)[ active_buf_id[0] ]);
|
|
||||||
drawenv_put (& r_(screen_buf->draw )[ active_buf_id[0] ]);
|
|
||||||
|
|
||||||
// Initialize and setup the GTE geometry offsets
|
|
||||||
geom_init();
|
|
||||||
geom_set_offset(ScreenRes_CenterX, ScreenRes_CenterY);
|
|
||||||
geom_set_screen(ScreenZ);
|
|
||||||
|
|
||||||
set_display_enabled(1); // gp_DisplayEnabled
|
|
||||||
}
|
|
||||||
|
|
||||||
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
|
||||||
draw_sync(0);
|
|
||||||
vsync(0);
|
|
||||||
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
|
||||||
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
|
||||||
{
|
|
||||||
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
|
||||||
pa->used = 0;
|
|
||||||
}
|
|
||||||
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
|
||||||
}
|
|
||||||
|
|
||||||
GCC_OPTIMIZATION_DISABLE
|
GCC_OPTIMIZATION_DISABLE
|
||||||
void update(PrimitiveArena* pa, U4* ordering_buf)
|
void update(PrimitiveArena* pa, U4* ordering_buf)
|
||||||
{
|
{
|
||||||
@@ -484,6 +382,19 @@ GCC_OPTIMIZATION_ENABLE
|
|||||||
void render(void) {
|
void render(void) {
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void gp_display_frame(DoubleBuffer* screen_buf, S4* active_buf_id, U4* ordering_buf, PrimitiveArena* pa) {
|
||||||
|
draw_sync(0);
|
||||||
|
vsync(0);
|
||||||
|
displayenv_put(& r_(screen_buf->display)[active_buf_id[0] ]);
|
||||||
|
drawenv_put (& r_(screen_buf->draw) [active_buf_id[0] ]);
|
||||||
|
{
|
||||||
|
draw_orderingtbl(ordering_buf + OrderingTbl_Len - 1);
|
||||||
|
pa->used = 0;
|
||||||
|
}
|
||||||
|
active_buf_id[0] = ! active_buf_id[0]; // Swap current buffer
|
||||||
|
}
|
||||||
|
|
||||||
|
GCC_OPTIMIZATION_DISABLE
|
||||||
int main(void)
|
int main(void)
|
||||||
{
|
{
|
||||||
smem = (SMemory){0};
|
smem = (SMemory){0};
|
||||||
@@ -527,3 +438,4 @@ int main(void)
|
|||||||
};
|
};
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
GCC_OPTIMIZATION_ENABLE
|
||||||
|
|||||||
@@ -26,3 +26,77 @@ enum {
|
|||||||
};
|
};
|
||||||
|
|
||||||
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
#define v3s4_fp_one() v3s4(fp_one, fp_one, fp_one)
|
||||||
|
|
||||||
|
typedef U4 OrderingTable_Buffer[OrderingTbl_Len];
|
||||||
|
typedef Array_(OrderingTable_Buffer, 2);
|
||||||
|
|
||||||
|
typedef B1 PrimitiveBuffer[PrimitiveBuff_Len];
|
||||||
|
typedef Array_(PrimitiveBuffer, 2);
|
||||||
|
typedef Struct_(PrimitiveArena) {
|
||||||
|
A2_PrimitiveBuffer buf;
|
||||||
|
U4 used;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Cube_num_verts 8
|
||||||
|
typedef Array_(V3_S2, Cube_num_verts);
|
||||||
|
#define Cube_num_faces 6
|
||||||
|
typedef Array_(V4_S2, Cube_num_faces);
|
||||||
|
I_ void ent_cube128_init(A8_V3_S2* verts, A6_V4_S2* faces) {
|
||||||
|
LP_ A8_V3_S2 baked_verts = (A8_V3_S2) {
|
||||||
|
{ -128, -128, -128 },
|
||||||
|
{ 128, -128, -128 },
|
||||||
|
{ 128, -128, 128 },
|
||||||
|
{ -128, -128, 128 },
|
||||||
|
{ -128, 128, -128 },
|
||||||
|
{ 128, 128, -128 },
|
||||||
|
{ 128, 128, 128 },
|
||||||
|
{ -128, 128, 128 }
|
||||||
|
};
|
||||||
|
LP_ A6_V4_S2 baked_faces = (A6_V4_S2) {
|
||||||
|
{ 3, 2, 0, 1 },
|
||||||
|
{ 0, 1, 4, 5 },
|
||||||
|
{ 4, 5, 7, 6 },
|
||||||
|
{ 1, 2, 5, 6 },
|
||||||
|
{ 2, 3, 6, 7 },
|
||||||
|
{ 3, 0, 7, 4 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A8_V3_S2) );
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A6_V4_S2) );
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
typedef Struct_(Ent_Cube) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 vel;
|
||||||
|
V3_S4 pos;
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A8_V3_S2 verts;
|
||||||
|
A6_V4_S2 faces;
|
||||||
|
};
|
||||||
|
|
||||||
|
#define Floor_num_verts 4
|
||||||
|
typedef Array_(V3_S2, Floor_num_verts);
|
||||||
|
#define Floor_num_faces 2
|
||||||
|
typedef Array_(V3_S2, Floor_num_faces);
|
||||||
|
I_ void ent_floor_init(A4_V3_S2* verts, A2_V3_S2* faces) {
|
||||||
|
LP_ A4_V3_S2 baked_verts = (A4_V3_S2) {
|
||||||
|
{ -900, 0, -900 },
|
||||||
|
{ -900, 0, 900 },
|
||||||
|
{ 900, 0, -900 },
|
||||||
|
{ 900, 0, 900 },
|
||||||
|
};
|
||||||
|
LP_ A2_V3_S2 baked_faces = (A2_V3_S2) {
|
||||||
|
{ 0, 1, 2 },
|
||||||
|
{ 1, 3, 2 },
|
||||||
|
};
|
||||||
|
mem_copy(u4_(verts), u4_(& baked_verts), S_(A4_V3_S2));
|
||||||
|
mem_copy(u4_(faces), u4_(& baked_faces), S_(A2_V3_S2));
|
||||||
|
};
|
||||||
|
typedef Struct_(Ent_Floor) {
|
||||||
|
V3_S4 accel;
|
||||||
|
V3_S4 pos;
|
||||||
|
V3_S4 scale;
|
||||||
|
V3_S2 rot;
|
||||||
|
A4_V3_S2 verts;
|
||||||
|
A2_V3_S2 faces;
|
||||||
|
};
|
||||||
|
|||||||
@@ -1,3 +0,0 @@
|
|||||||
#ifdef INTELLISENSE_DIRECTIVES
|
|
||||||
# include "psyq.h"
|
|
||||||
#endif
|
|
||||||
@@ -24,8 +24,8 @@
|
|||||||
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
* Emits 9 instructions (status/buttons/axes/attempt stores plus the
|
||||||
* two-instruction zero-extended buttons load).
|
* two-instruction zero-extended buttons load).
|
||||||
*/
|
*/
|
||||||
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
FI_ Slice_MipsCode ac_pad_sio_write_pad_state(MipsAtomBuilder_R ab, U4 status_val, U4 state_ptr_reg, U4 scratch_reg)
|
||||||
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, {
|
MipsAtomComp_Proc_(ac_pad_sio_write_pad_state, ab, {
|
||||||
add_ui(scratch_reg, R_0, status_val),
|
add_ui(scratch_reg, R_0, status_val),
|
||||||
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
store_word(scratch_reg, state_ptr_reg, O_(PadState,status)),
|
||||||
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
/* FIX 2026-08-02: buttons = 0x0000FFFF = "no buttons pressed" in
|
||||||
|
|||||||
+10625
File diff suppressed because one or more lines are too long
+125
-120
@@ -180,12 +180,9 @@ function link-modules { param([string[]]$link_modules, [string] $elf, [string[]
|
|||||||
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
$link_args += ($f_link_pass_through_prefix + $f_link_mapfile + $map)
|
||||||
|
|
||||||
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
$link_args += ($f_link_pass_through_prefix + $f_link_start_group)
|
||||||
# raw_sio_pad_poll_20260802 — Task 5.1c surgical library-list trim.
|
# 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math, mcrd, mcx, press, sio, snd, spu, tap)
|
||||||
# The 16 removed entries (c2, card, cd, comb, ds, gs, gun, hmd, math,
|
# had LOAD lines in the map but ZERO .o files pulled in — they were unused.
|
||||||
# mcrd, mcx, press, sio, snd, spu, tap) had LOAD lines in the map but
|
# 5 kept libraries (api, c, etc, gpu, gte) are required by the C-side calls in hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
||||||
# ZERO .o files pulled in — they were unused. The 5 kept libraries
|
|
||||||
# (api, c, etc, gpu, gte) are required by the C-side calls in
|
|
||||||
# hello_joypad.c (reset_graph, draw_sync, vsync, etc.).
|
|
||||||
$libraries = @(
|
$libraries = @(
|
||||||
"api",
|
"api",
|
||||||
"c",
|
"c",
|
||||||
@@ -227,9 +224,7 @@ function ps1-meta { param(
|
|||||||
[string[]]$passes = @('--pre-link'),
|
[string[]]$passes = @('--pre-link'),
|
||||||
[string[]]$extra_args = @()
|
[string[]]$extra_args = @()
|
||||||
)
|
)
|
||||||
# `--unity-root` and `--source` are
|
# `--unity-root` and `--source` are mutually exclusive. Exactly one of `$unity_root` / `$sources` must be supplied; the other must be absent.
|
||||||
# mutually exclusive. Exactly one of `$unity_root` / `$sources` must
|
|
||||||
# be supplied; the other must be absent.
|
|
||||||
if ($null -ne $unity_root -and $unity_root -ne '')
|
if ($null -ne $unity_root -and $unity_root -ne '')
|
||||||
{
|
{
|
||||||
if ($null -ne $sources -and $sources.Count -gt 0) {
|
if ($null -ne $sources -and $sources.Count -gt 0) {
|
||||||
@@ -265,6 +260,73 @@ function ps1-meta { param(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function inject-dwarf { param(
|
||||||
|
[string]$elf,
|
||||||
|
[string]$path_gen
|
||||||
|
)
|
||||||
|
$base_name = [System.IO.Path]::GetFileNameWithoutExtension($elf)
|
||||||
|
$path_dwarf_line_bin = join-path $path_gen "$base_name.dwarf_line.bin"
|
||||||
|
$path_dwarf_aranges_bin = join-path $path_gen "$base_name.dwarf_aranges.bin"
|
||||||
|
$path_dwarf_rnglists_bin = join-path $path_gen "$base_name.dwarf_rnglists.bin"
|
||||||
|
$path_dwarf_info_bin = join-path $path_gen "$base_name.dwarf_info.bin"
|
||||||
|
$path_dwarf_abbrev_bin = join-path $path_gen "$base_name.dwarf_abbrev.bin"
|
||||||
|
$path_dwarf_str_bin = join-path $path_gen "$base_name.dwarf_str.bin"
|
||||||
|
$path_dwarf_loc_bin = join-path $path_gen "$base_name.dwarf_loc.bin"
|
||||||
|
$path_dwarf_loclists_bin = join-path $path_gen "$base_name.dwarf_loclists.bin"
|
||||||
|
$path_inject_elf = join-path $path_build "$base_name.dwarf-injected.elf"
|
||||||
|
|
||||||
|
if (-not (Test-Path $path_dwarf_line_bin)) { return }
|
||||||
|
if (-not (Test-Path $path_dwarf_aranges_bin)) { return }
|
||||||
|
if (-not (Test-Path $path_dwarf_rnglists_bin)) { return }
|
||||||
|
|
||||||
|
Write-Host "[build] DWARF-injecting $elf -> $path_inject_elf"
|
||||||
|
Copy-Item -LiteralPath $elf -Destination $path_inject_elf -Force
|
||||||
|
|
||||||
|
# Objcopy call 1: 3x --update-section for the PC-mapping tables (line, aranges, rnglists).
|
||||||
|
$objcopy_args_dwarf_pc = @(
|
||||||
|
"--update-section=.debug_line=$path_dwarf_line_bin",
|
||||||
|
"--update-section=.debug_aranges=$path_dwarf_aranges_bin",
|
||||||
|
"--update-section=.debug_rnglists=$path_dwarf_rnglists_bin"
|
||||||
|
)
|
||||||
|
& $Objcopy @objcopy_args_dwarf_pc $path_inject_elf 2>&1 | Out-Null
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
Write-Warning "[build] objcopy dwarf-pc splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||||
|
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
# Objcopy call 2: 3x --update-section + 2x --add-section for the debug-data tables (info, abbrev, str, loc, loclists).
|
||||||
|
$objcopy_args_dwarf_info = @(
|
||||||
|
"--update-section=.debug_info=$path_dwarf_info_bin",
|
||||||
|
"--update-section=.debug_abbrev=$path_dwarf_abbrev_bin",
|
||||||
|
"--update-section=.debug_str=$path_dwarf_str_bin",
|
||||||
|
"--add-section=.debug_loc=$path_dwarf_loc_bin",
|
||||||
|
"--add-section=.debug_loclists=$path_dwarf_loclists_bin"
|
||||||
|
)
|
||||||
|
& $Objcopy @objcopy_args_dwarf_info $path_inject_elf 2>&1 | Out-Null
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
Write-Warning "[build] objcopy dwarf-info splice failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||||
|
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
||||||
|
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
||||||
|
# The original ELF and PS-EXE remain byte/flag unchanged.
|
||||||
|
& $Objcopy `
|
||||||
|
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
||||||
|
--set-section-flags ".data=alloc,load,data,code,contents" `
|
||||||
|
$path_inject_elf 2>&1 | Out-Null
|
||||||
|
if ($LASTEXITCODE -ne 0) {
|
||||||
|
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $path_inject_elf"
|
||||||
|
Remove-Item -LiteralPath $path_inject_elf -ErrorAction SilentlyContinue
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
Write-Host "[build] DWARF-injected ELF: $path_inject_elf"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
# inject-dwarf
|
||||||
|
|
||||||
function build-hello_psyqo {
|
function build-hello_psyqo {
|
||||||
$includes += @()
|
$includes += @()
|
||||||
|
|
||||||
@@ -391,61 +453,7 @@ function build-hello_gte {
|
|||||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
$dwarfLineBin = join-path $path_build_gen 'hello_gte.dwarf_line.bin'
|
inject-dwarf $elf $path_build_gen
|
||||||
$dwarfArangesBin = join-path $path_build_gen 'hello_gte.dwarf_aranges.bin'
|
|
||||||
$dwarfRnglistsBin = join-path $path_build_gen 'hello_gte.dwarf_rnglists.bin'
|
|
||||||
$injectElf = join-path $path_build 'hello_gte.dwarf-injected.elf'
|
|
||||||
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
|
|
||||||
{
|
|
||||||
Write-Host "[build] DWARF-injecting $elf -> $injectElf"
|
|
||||||
Copy-Item -LiteralPath $elf -Destination $injectElf -Force
|
|
||||||
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
|
|
||||||
$f_args = @(
|
|
||||||
"--update-section=.debug_line=$dwarfLineBin",
|
|
||||||
"--update-section=.debug_aranges=$dwarfArangesBin",
|
|
||||||
"--update-section=.debug_rnglists=$dwarfRnglistsBin"
|
|
||||||
)
|
|
||||||
& $Objcopy @f_args $injectElf 2>&1 | Out-Null
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
Write-Warning "[build] objcopy F' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
|
||||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
$dwarfInfoBin = join-path $path_build_gen 'hello_gte.dwarf_info.bin'
|
|
||||||
$dwarfAbbrevBin = join-path $path_build_gen 'hello_gte.dwarf_abbrev.bin'
|
|
||||||
$dwarfStrBin = join-path $path_build_gen 'hello_gte.dwarf_str.bin'
|
|
||||||
$dwarfLocBin = join-path $path_build_gen 'hello_gte.dwarf_loc.bin'
|
|
||||||
$dwarfLoclistsBin = join-path $path_build_gen 'hello_gte.dwarf_loclists.bin'
|
|
||||||
$g_args = @(
|
|
||||||
"--update-section=.debug_info=$dwarfInfoBin",
|
|
||||||
"--update-section=.debug_abbrev=$dwarfAbbrevBin",
|
|
||||||
"--update-section=.debug_str=$dwarfStrBin",
|
|
||||||
"--add-section=.debug_loc=$dwarfLocBin",
|
|
||||||
"--add-section=.debug_loclists=$dwarfLoclistsBin"
|
|
||||||
)
|
|
||||||
& $Objcopy @g_args $injectElf 2>&1 | Out-Null
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
Write-Warning "[build] objcopy G' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
|
||||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
|
||||||
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
|
||||||
# The original ELF and PS-EXE remain byte/flag unchanged.
|
|
||||||
& $Objcopy `
|
|
||||||
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
|
||||||
--set-section-flags ".data=alloc,load,data,code,contents" `
|
|
||||||
$injectElf 2>&1 | Out-Null
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
|
|
||||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
Write-Host "[build] DWARF-injected ELF: $injectElf"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
# build-hello_gte
|
# build-hello_gte
|
||||||
|
|
||||||
@@ -496,63 +504,60 @@ function build-hello_joypad {
|
|||||||
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
# Post-link: gdb-runtime + dwarf-injection in a single Lua invocation (one luajit cold start).
|
||||||
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
$dwarfLineBin = join-path $path_build_gen 'hello_joypad.dwarf_line.bin'
|
inject-dwarf $elf $path_build_gen
|
||||||
$dwarfArangesBin = join-path $path_build_gen 'hello_joypad.dwarf_aranges.bin'
|
|
||||||
$dwarfRnglistsBin = join-path $path_build_gen 'hello_joypad.dwarf_rnglists.bin'
|
|
||||||
$injectElf = join-path $path_build 'hello_joypad.dwarf-injected.elf'
|
|
||||||
if ((Test-Path $dwarfLineBin) -and (Test-Path $dwarfArangesBin) -and (Test-Path $dwarfRnglistsBin))
|
|
||||||
{
|
|
||||||
Write-Host "[build] DWARF-injecting $elf -> $injectElf"
|
|
||||||
Copy-Item -LiteralPath $elf -Destination $injectElf -Force
|
|
||||||
# Objcopy call: 3x --update-section for (line, aranges, rnglists).
|
|
||||||
$f_args = @(
|
|
||||||
"--update-section=.debug_line=$dwarfLineBin",
|
|
||||||
"--update-section=.debug_aranges=$dwarfArangesBin",
|
|
||||||
"--update-section=.debug_rnglists=$dwarfRnglistsBin"
|
|
||||||
)
|
|
||||||
& $Objcopy @f_args $injectElf 2>&1 | Out-Null
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
Write-Warning "[build] objcopy F' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
|
||||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
$dwarfInfoBin = join-path $path_build_gen 'hello_joypad.dwarf_info.bin'
|
|
||||||
$dwarfAbbrevBin = join-path $path_build_gen 'hello_joypad.dwarf_abbrev.bin'
|
|
||||||
$dwarfStrBin = join-path $path_build_gen 'hello_joypad.dwarf_str.bin'
|
|
||||||
$dwarfLocBin = join-path $path_build_gen 'hello_joypad.dwarf_loc.bin'
|
|
||||||
$dwarfLoclistsBin = join-path $path_build_gen 'hello_joypad.dwarf_loclists.bin'
|
|
||||||
$g_args = @(
|
|
||||||
"--update-section=.debug_info=$dwarfInfoBin",
|
|
||||||
"--update-section=.debug_abbrev=$dwarfAbbrevBin",
|
|
||||||
"--update-section=.debug_str=$dwarfStrBin",
|
|
||||||
"--add-section=.debug_loc=$dwarfLocBin",
|
|
||||||
"--add-section=.debug_loclists=$dwarfLoclistsBin"
|
|
||||||
)
|
|
||||||
& $Objcopy @g_args $injectElf 2>&1 | Out-Null
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
Write-Warning "[build] objcopy G' splice failed (exit $LASTEXITCODE); removing $injectElf"
|
|
||||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
# Baked atoms execute from RAM but are emitted as C data arrays, so their ELF sections lack SHF_EXECINSTR.
|
|
||||||
# GDB discards line rows for non-code sections. Mark only the debug-copy sections executable.
|
|
||||||
# The original ELF and PS-EXE remain byte/flag unchanged.
|
|
||||||
& $Objcopy `
|
|
||||||
--set-section-flags ".rodata=alloc,load,readonly,code,contents" `
|
|
||||||
--set-section-flags ".data=alloc,load,data,code,contents" `
|
|
||||||
$injectElf 2>&1 | Out-Null
|
|
||||||
if ($LASTEXITCODE -ne 0) {
|
|
||||||
Write-Warning "[build] atom-section flag update failed (exit $LASTEXITCODE); removing $injectElf"
|
|
||||||
Remove-Item -LiteralPath $injectElf -ErrorAction SilentlyContinue
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
Write-Host "[build] DWARF-injected ELF: $injectElf"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
build-hello_joypad
|
# build-hello_joypad
|
||||||
|
|
||||||
|
function build-hello_camera {
|
||||||
|
$includes += @()
|
||||||
|
|
||||||
|
$path_module = join-path $path_code 'hello_camera'
|
||||||
|
$path_duffle = join-path $path_code 'duffle'
|
||||||
|
$path_atom_metadata = join-path $path_duffle 'word_count.metadata.h'
|
||||||
|
$path_build_gen = join-path $path_build 'gen'
|
||||||
|
|
||||||
|
$src_c = join-path $path_module 'hello_camera.c'
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--pre-link')
|
||||||
|
|
||||||
|
$assemble_args = @()
|
||||||
|
$assemble_args += $f_debug
|
||||||
|
$assemble_args += $f_optimize_none
|
||||||
|
$assemble_args += ($f_include + $path_code)
|
||||||
|
|
||||||
|
$src_asm_crt = join-path $path_nugget_common 'crt0/crt0.s'
|
||||||
|
$module_asm_crt = join-path $path_build 'crt0.o'
|
||||||
|
assemble-unit $src_asm_crt $module_asm_crt $includes $assemble_args
|
||||||
|
|
||||||
|
$module_c = join-path $path_build 'hello_camera_c.o'
|
||||||
|
|
||||||
|
$compile_args = @()
|
||||||
|
$compile_args += $f_debug
|
||||||
|
$compile_args += ($f_define + 'BUILD_DEBUG')
|
||||||
|
$compile_args += $f_optimize_none
|
||||||
|
# $compile_args += $f_optimize_intrinsics
|
||||||
|
# $compile_args += $f_optimize_size
|
||||||
|
# $compile_args += $f_optimize_debug
|
||||||
|
$compile_args += ($f_include + $path_code)
|
||||||
|
compile-unit $src_c $module_c $includes $compile_args
|
||||||
|
|
||||||
|
$elf = join-path $path_build 'hello_camera.elf'
|
||||||
|
$exe = join-path $path_build 'hello_camera.ps-exe'
|
||||||
|
|
||||||
|
$link_args = @()
|
||||||
|
$link_args += $f_debug
|
||||||
|
# $link_args += $f_optimize_size
|
||||||
|
$link_modules = @(
|
||||||
|
$module_asm_crt,
|
||||||
|
$module_c
|
||||||
|
)
|
||||||
|
link-modules $link_modules $elf $link_args
|
||||||
|
make-binary $elf $exe
|
||||||
|
|
||||||
|
ps1-meta -unity_root $src_c -metadata $path_atom_metadata -out_root $path_build_gen -passes @('--post-link') ` -extra_args @('--elf', $elf)
|
||||||
|
|
||||||
|
inject-dwarf $elf $path_build_gen
|
||||||
|
}
|
||||||
|
build-hello_camera
|
||||||
|
|
||||||
# NO idea if this works yet...
|
# NO idea if this works yet...
|
||||||
function Send-ToEmulator { param( [string]$exePath )
|
function Send-ToEmulator { param( [string]$exePath )
|
||||||
|
|||||||
+409
-192
File diff suppressed because it is too large
Load Diff
@@ -47,15 +47,14 @@ local function find_repo_root()
|
|||||||
return root
|
return root
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and
|
--- Set `package.path` (for `require("duffle")` + `require("passes.X")`) and `package.cpath` (for `lpeg.dll`).
|
||||||
--- `package.cpath` (for `lpeg.dll`).
|
|
||||||
---
|
---
|
||||||
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
--- This script does NOT touch the OS environment: no `os.setenv`, no `os.putenv`, no `$PATH` mods.
|
||||||
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
--- It just sets `package.path` and `package.cpath` (the standard Lua way to register module search dirs).
|
||||||
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
|
--- lpeg is built by `update_deps.ps1` to `toolchain/lpeg/`,
|
||||||
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
--- which we wire into `package.cpath` here (so `require("lpeg")` from `duffle.lua` resolves without any global state).
|
||||||
function M.setup()
|
function M.setup()
|
||||||
local repo_root = find_repo_root()
|
local repo_root = find_repo_root()
|
||||||
if not repo_root then
|
if not repo_root then
|
||||||
-- Unreachable in practice: find_repo_root() derives the repo root from this script's
|
-- Unreachable in practice: find_repo_root() derives the repo root from this script's
|
||||||
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
-- own source path via debug.getinfo(1, "S").source (no subprocess, no git CLI, <1ms).
|
||||||
|
|||||||
@@ -0,0 +1,418 @@
|
|||||||
|
-- elf32.lua — Pure-Lua ELF32 format helpers with no lfs / no lpeg dependency.
|
||||||
|
-- The reload helper's `parse_manifest` (scripts/pcsx_debug_helper/reload.lua)
|
||||||
|
-- and the metaprogram's `read_elf_sections` + `read_nm` (scripts/elf_dwarf.lua)
|
||||||
|
-- both parsed ELF32 headers from wire bytes.
|
||||||
|
--
|
||||||
|
-- This module contains the format constants and the byte-level walker.
|
||||||
|
--- The metaprogram side keeps `read_u32_le` / `read_u16_le` as local forwarders; the helper side calls `E.*` directly.
|
||||||
|
--
|
||||||
|
-- **Adapter contract (explicit pass style):**
|
||||||
|
-- The helper VM's `Support.File` exposes byte-read methods that require `self` (fileffi.lua:225-227),
|
||||||
|
-- so callers wrap once in a 1-line adapter that strips `self`.
|
||||||
|
-- The parsers here operate on the unwrapped form.
|
||||||
|
-- Reads are flat function calls — `E.read_u8(adapter, off)`, `E.read_u32(adapter, off)`, `E.size(adapter)`.
|
||||||
|
-- read_u8(adapter, off) -> integer | nil
|
||||||
|
-- read_u16(adapter, off) -> integer | nil
|
||||||
|
-- read_u32(adapter, off) -> integer | nil
|
||||||
|
-- size(adapter) -> integer
|
||||||
|
--
|
||||||
|
-- **Convention:** every offset in the constants tables is a zero-based wire offset.
|
||||||
|
-- The `+ 1` conversion happens only at the `string.byte` boundary inside the readers.
|
||||||
|
--
|
||||||
|
-- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
||||||
|
-- spec: System V ABI gABI v1.2 §"Symbol Table" (Elf32_Sym layout)
|
||||||
|
|
||||||
|
local M = {}
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Little-endian readers (bit-weighted accumulator, math.floor only)
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Read a 4-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||||
|
---
|
||||||
|
--- Bit weights are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
||||||
|
--- byte 0 contributes its value directly;
|
||||||
|
--- byte 1 is shifted left by 8; byte 2 by 16; byte 3 by 24.
|
||||||
|
---
|
||||||
|
--- math.floor (not LuaJIT's `>>`) keeps the body portable across LuaJIT 2.0/2.1 and plain Lua 5.x. `string.byte` receives `+ 1` at the boundary.
|
||||||
|
---
|
||||||
|
--- **Call form:** explicit-pass. The reader receives `adapter` as the first positional argument and the offset as the second; no `self` is passed.
|
||||||
|
--- Test fixtures declare `function(offset) ... end` and the parsers call them via dot syntax `adapter.read_u8_at(off)`.
|
||||||
|
--- The colon form `adapter:read_u8_at(off)` would prepend the adapter table as `offset` and break the contract.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer|nil
|
||||||
|
function M.read_u32(adapter, off)
|
||||||
|
return adapter.read_u8_at(off)
|
||||||
|
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||||
|
+ adapter.read_u8_at(off + 0x02) * 0x00010000
|
||||||
|
+ adapter.read_u8_at(off + 0x03) * 0x01000000
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read a 2-byte little-endian unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer|nil
|
||||||
|
function M.read_u16(adapter, off)
|
||||||
|
return adapter.read_u8_at(off)
|
||||||
|
+ adapter.read_u8_at(off + 0x01) * 0x00000100
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read a 1-byte unsigned integer from `adapter` at zero-based wire offset `off`.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer|nil
|
||||||
|
function M.read_u8(adapter, off)
|
||||||
|
return adapter.read_u8_at(off)
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Total adapter byte length.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @return integer
|
||||||
|
function M.size(adapter)
|
||||||
|
return adapter.read_size()
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Forwarders kept for backward compat with scripts/elf_dwarf.lua.
|
||||||
|
--- The metaprogram side keeps `read_u32_le` / `read_u16_le`;
|
||||||
|
--- both layers now use the same byte-level helpers under the hood.
|
||||||
|
function M.read_u32_le(buf, off)
|
||||||
|
local byte_off = off + 1
|
||||||
|
return buf:byte(byte_off)
|
||||||
|
+ buf:byte(byte_off + 0x01) * 0x00000100
|
||||||
|
+ buf:byte(byte_off + 0x02) * 0x00010000
|
||||||
|
+ buf:byte(byte_off + 0x03) * 0x01000000
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||||
|
--- @param buf string
|
||||||
|
--- @param off integer -- zero-based wire offset
|
||||||
|
--- @return integer
|
||||||
|
function M.read_u16_le(buf, off)
|
||||||
|
local byte_off = off + 1
|
||||||
|
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Format constants
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
-- ELF format constants (System V ABI gABI v1.2).
|
||||||
|
M.ELFCLASS32 = 1 -- spec: gABI v1.2 §"ELF Header" — EI_CLASS byte
|
||||||
|
M.ELFDATA2LSB = 1 -- spec: gABI v1.2 §"ELF Header" — EI_DATA byte
|
||||||
|
M.EM_MIPS = 8 -- spec: gABI v1.2 §"Machine Information" — MIPS architecture
|
||||||
|
|
||||||
|
-- Section type constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||||
|
M.SHT_SYMTAB = 2 -- spec: gABI v1.2 §"Section Types" — symbol table
|
||||||
|
M.SHT_STRTAB = 3 -- spec: gABI v1.2 §"Section Types" — string table
|
||||||
|
M.SHT_NOBITS = 8 -- spec: gABI v1.2 §"Section Types" — no space in file
|
||||||
|
|
||||||
|
-- Section flag constants (System V ABI gABI v1.2 §"Section Header Table").
|
||||||
|
M.SHF_WRITE = 0x1 -- spec: gABI v1.2 §"Section Attributes" — writable
|
||||||
|
M.SHF_ALLOC = 0x2 -- spec: gABI v1.2 §"Section Attributes" — occupies memory
|
||||||
|
M.SHF_EXECINSTR = 0x4 -- spec: gABI v1.2 §"Section Attributes" — executable
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 header layout (System V ABI gABI v1.2 §"ELF Header" Table 1)
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- All offsets are zero-based wire offsets. The header is 52 bytes total (header_bytes = 0x34 = 52).
|
||||||
|
M.ELF32_HEADER = {
|
||||||
|
magic_offset = 0x00, -- 4 bytes; expected "\127ELF"
|
||||||
|
magic = "\127ELF",
|
||||||
|
class_offset = 0x04, -- 1 byte; 1 = ELF32, 2 = ELF64
|
||||||
|
endian_offset = 0x05, -- 1 byte; 1 = little-endian, 2 = big-endian
|
||||||
|
header_bytes = 0x34, -- ELF32 header is 52 bytes total
|
||||||
|
e_entry_offset = 0x18, -- 4-byte LE; entry-point virtual address
|
||||||
|
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
||||||
|
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
||||||
|
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
||||||
|
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 section-header layout (System V ABI gABI v1.2 §"Section Header Table")
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- Each entry is 40 bytes (sh_entsize_bytes = 0x28 = 40);
|
||||||
|
-- zero-based, field offsets relative to the start of the entry.
|
||||||
|
M.ELF32_SECTION = {
|
||||||
|
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
||||||
|
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
||||||
|
sh_flags_offset = 0x08, -- 4-byte LE; section flags (SHF_*)
|
||||||
|
sh_addr_offset = 0x0C, -- 4-byte LE; virtual address at execution
|
||||||
|
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
||||||
|
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
||||||
|
sh_link_offset = 0x18, -- 4-byte LE; link to a related section
|
||||||
|
sh_entsize_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — 40 bytes per entry
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- ELF32 symbol-table entry layout (System V ABI gABI v1.2 §"Symbol Table")
|
||||||
|
-- ---------------------------------------------------------------------------
|
||||||
|
-- Each entry is 16 bytes (sym_entry_bytes = 0x10 = 16);
|
||||||
|
-- zero-based, field offsets relative to the start of the entry.
|
||||||
|
M.ELF32_SYM = {
|
||||||
|
st_name = 0x00, -- 4-byte LE; offset into the linked string table
|
||||||
|
st_value = 0x04, -- 4-byte LE; symbol value (address / absolute)
|
||||||
|
st_size = 0x08, -- 4-byte LE; symbol size in bytes
|
||||||
|
st_info = 0x0C, -- 1 byte; binding (high nibble) + type (low nibble)
|
||||||
|
sym_entry_bytes = 0x10, -- spec: gABI v1.2 §"Symbol Table" — 16 bytes per entry
|
||||||
|
}
|
||||||
|
|
||||||
|
-- DWARF32 initial-length terminator (DWARF4 §7.4) — kept here so the metaprogram's elf_dwarf.lua can drop its own copy of the same constant.
|
||||||
|
M.dw_dwarf32_terminator = 0xFFFFFFFF
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Adapter validation
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Validate that `adapter` exposes the byte-read surface.
|
||||||
|
--- Returns true on success, false + a stable error code on failure.
|
||||||
|
--- The helper side calls this before parse_manifest to reject callers before any byte is read.
|
||||||
|
--- @param adapter any
|
||||||
|
--- @return boolean, string|nil
|
||||||
|
function M.validate_adapter(adapter)
|
||||||
|
if type(adapter) ~= "table" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_u8_at) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_u16_at) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_u32_at) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
if type(adapter.read_size) ~= "function" then return false, "bad_file_adapter" end
|
||||||
|
return true, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- String-table reader
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Extract a NUL-terminated C string from `strtab` at zero-based offset `off`.
|
||||||
|
--- Returns nil if `off` is out of range or the string is not NUL-terminated.
|
||||||
|
--- @param strtab string
|
||||||
|
--- @param off integer
|
||||||
|
--- @return string|nil
|
||||||
|
function M.get_str(strtab, off)
|
||||||
|
if off < 0 or off >= #strtab then return nil end
|
||||||
|
local end_pos = strtab:find("\0", off + 1, true)
|
||||||
|
if not end_pos then return nil end
|
||||||
|
return strtab:sub(off + 1, end_pos - 1)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Header / section / symbol walkers
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Read the ELF32 header through `adapter` and validate the magic, class, and data encoding.
|
||||||
|
--- Returns a table on success:
|
||||||
|
--- { e_entry, e_shoff, e_shentsize, e_shnum, e_shstrndx, error = nil }
|
||||||
|
--- On failure returns nil + a stable error code:
|
||||||
|
--- bad_magic, unsupported_elf_class, unsupported_elf_data, truncated_header
|
||||||
|
--- The header's machine field is NOT validated here — callers (e.g. the helper's prime path) decide whether to require EM_MIPS before symbol reads.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @return table|nil, string|nil
|
||||||
|
function M.parse_elf32_headers(adapter)
|
||||||
|
local ok, err = M.validate_adapter(adapter)
|
||||||
|
if not ok then return nil, err end
|
||||||
|
|
||||||
|
-- 4-byte magic: 0x7F 'E' 'L' 'F'.
|
||||||
|
-- The byte readers take the adapter explicitly.
|
||||||
|
-- The production `Support.File` adapter is wrapped by the caller to drop its implicit `self` so the parser shape is flat pass-style.
|
||||||
|
local b1 = M.read_u8(adapter, 0)
|
||||||
|
local b2 = M.read_u8(adapter, 1)
|
||||||
|
local b3 = M.read_u8(adapter, 2)
|
||||||
|
local b4 = M.read_u8(adapter, 3)
|
||||||
|
if not (b1 and b2 and b3 and b4)
|
||||||
|
or not (b1 == 0x7f and b2 == 0x45 and b3 == 0x4c and b4 == 0x46) then
|
||||||
|
return nil, "bad_magic"
|
||||||
|
end
|
||||||
|
|
||||||
|
local class = M.read_u8(adapter, M.ELF32_HEADER.class_offset)
|
||||||
|
if class ~= M.ELFCLASS32 then
|
||||||
|
return nil, "unsupported_elf_class"
|
||||||
|
end
|
||||||
|
|
||||||
|
local data = M.read_u8(adapter, M.ELF32_HEADER.endian_offset)
|
||||||
|
if data ~= M.ELFDATA2LSB then
|
||||||
|
return nil, "unsupported_elf_data"
|
||||||
|
end
|
||||||
|
|
||||||
|
local e_entry = M.read_u32(adapter, M.ELF32_HEADER.e_entry_offset)
|
||||||
|
local e_shoff = M.read_u32(adapter, M.ELF32_HEADER.e_shoff_offset)
|
||||||
|
local e_shentsize = M.read_u16(adapter, M.ELF32_HEADER.e_shentsize_offset)
|
||||||
|
local e_shnum = M.read_u16(adapter, M.ELF32_HEADER.e_shnum_offset)
|
||||||
|
local e_shstrndx = M.read_u16(adapter, M.ELF32_HEADER.e_shstrndx_offset)
|
||||||
|
if not (e_entry and e_shoff and e_shentsize and e_shnum and e_shstrndx) then
|
||||||
|
return nil, "truncated_header"
|
||||||
|
end
|
||||||
|
|
||||||
|
return {
|
||||||
|
e_entry = e_entry,
|
||||||
|
e_shoff = e_shoff,
|
||||||
|
e_shentsize = e_shentsize,
|
||||||
|
e_shnum = e_shnum,
|
||||||
|
e_shstrndx = e_shstrndx,
|
||||||
|
error = nil,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read one section-header entry from `adapter` at `sh_off`.
|
||||||
|
--- Returns a table with the wire fields plus a (yet-unresolved) `name` field.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param sh_off integer
|
||||||
|
--- @return table|nil, string|nil -- entry, error
|
||||||
|
local function read_section_entry(adapter, sh_off)
|
||||||
|
local entry = {
|
||||||
|
sh_name = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_name_offset),
|
||||||
|
sh_type = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_type_offset),
|
||||||
|
sh_flags = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_flags_offset),
|
||||||
|
sh_addr = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_addr_offset),
|
||||||
|
sh_offset = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_offset_offset),
|
||||||
|
sh_size = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_size_offset),
|
||||||
|
sh_link = M.read_u32(adapter, sh_off + M.ELF32_SECTION.sh_link_offset),
|
||||||
|
name = "",
|
||||||
|
}
|
||||||
|
if not (entry.sh_name and entry.sh_type and entry.sh_flags and entry.sh_addr
|
||||||
|
and entry.sh_offset and entry.sh_size and entry.sh_link) then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
return entry, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Walk every section header in `hdr` and return a 1-based array of entries
|
||||||
|
--- (the section at logical index 0 is at array position 1, etc.).
|
||||||
|
--- Each entry has the wire fields plus a resolved `name` derived from `.shstrtab`.
|
||||||
|
--- Returns nil + a stable error code on failure: truncated_section_headers, missing_shstrtab, truncated_strtab
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param hdr table -- the table returned by parse_elf32_headers
|
||||||
|
--- @return table|nil, string|nil
|
||||||
|
function M.walk_sections(adapter, hdr)
|
||||||
|
if not hdr or hdr.error then return nil, hdr and hdr.error or "truncated_section_headers" end
|
||||||
|
|
||||||
|
local file_size = M.size(adapter)
|
||||||
|
if hdr.e_shoff + hdr.e_shnum * hdr.e_shentsize > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Read every section header first; we need .shstrtab to resolve names.
|
||||||
|
local sections = {}
|
||||||
|
for i = 0, hdr.e_shnum - 1 do
|
||||||
|
local sh_off = hdr.e_shoff + i * hdr.e_shentsize
|
||||||
|
local entry, err = read_section_entry(adapter, sh_off)
|
||||||
|
if not entry then return nil, err end
|
||||||
|
sections[i + 1] = entry
|
||||||
|
end
|
||||||
|
|
||||||
|
if hdr.e_shstrndx >= hdr.e_shnum then
|
||||||
|
return nil, "missing_shstrtab"
|
||||||
|
end
|
||||||
|
|
||||||
|
local shstrtab = sections[hdr.e_shstrndx + 1]
|
||||||
|
if not shstrtab or shstrtab.sh_type ~= M.SHT_STRTAB then
|
||||||
|
return nil, "missing_shstrtab"
|
||||||
|
end
|
||||||
|
if shstrtab.sh_offset + shstrtab.sh_size > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local shstrtab_bytes = M.read_section_bytes(adapter, shstrtab)
|
||||||
|
if not shstrtab_bytes then return nil, "truncated_section_headers" end
|
||||||
|
|
||||||
|
for _, s in ipairs(sections) do
|
||||||
|
s.name = M.get_str(shstrtab_bytes, s.sh_name) or ""
|
||||||
|
end
|
||||||
|
|
||||||
|
return sections, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Read the bytes of one section. Returns a string, or nil if the adapter returns nil for any byte (out-of-bounds).
|
||||||
|
--- The caller is responsible fors sizing the buffer (the section's sh_offset + sh_size must fit in adapter.size).
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param section table -- one entry from walk_sections
|
||||||
|
--- @return string|nil
|
||||||
|
function M.read_section_bytes(adapter, section)
|
||||||
|
local size = section.sh_size
|
||||||
|
if size == 0 then return "" end
|
||||||
|
local out = {}
|
||||||
|
for i = 0, size - 1 do
|
||||||
|
local b = M.read_u8(adapter, section.sh_offset + i)
|
||||||
|
if b == nil then return nil end
|
||||||
|
out[#out + 1] = string.char(b)
|
||||||
|
end
|
||||||
|
return table.concat(out)
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Convenience: walk sections, then look up the named section, then read its bytes.
|
||||||
|
--- Returns nil + a stable error code if the section is absent or out-of-bounds.
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param sections table -- 1-based array from walk_sections
|
||||||
|
--- @param name string
|
||||||
|
--- @return string|nil, string|nil
|
||||||
|
function M.read_named_section(adapter, sections, name)
|
||||||
|
if not sections then return nil, "missing_section" end
|
||||||
|
for _, s in ipairs(sections) do
|
||||||
|
if s.name == name then
|
||||||
|
local bytes = M.read_section_bytes(adapter, s)
|
||||||
|
if not bytes then return nil, "truncated_section_data" end
|
||||||
|
return bytes, nil
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil, "missing_section"
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Walk every SHT_SYMTAB section in `sections` and accumulate symbols by name.
|
||||||
|
--- Each stored entry is `{ value = st_value, size = st_size, info = st_info, shndx = st_shndx }`.
|
||||||
|
--- Both STB_LOCAL and STB_GLOBAL symbols are included; the live ELF stores `smem` as a local symbol.
|
||||||
|
--- Returns nil + a stable error code on failure: missing_symtab_strtab, truncated_section_headers
|
||||||
|
--- @param adapter table
|
||||||
|
--- @param sections table
|
||||||
|
--- @return table|nil, string|nil
|
||||||
|
function M.collect_symbols(adapter, sections)
|
||||||
|
if not sections then return nil, "missing_sections" end
|
||||||
|
local symbols = {}
|
||||||
|
local file_size = M.size(adapter)
|
||||||
|
for _, s in ipairs(sections) do
|
||||||
|
if s.sh_type == M.SHT_SYMTAB then
|
||||||
|
local strtab = sections[s.sh_link + 1]
|
||||||
|
if not strtab or strtab.sh_type ~= M.SHT_STRTAB then
|
||||||
|
return nil, "missing_symtab_strtab"
|
||||||
|
end
|
||||||
|
if strtab.sh_offset + strtab.sh_size > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local strtab_bytes = M.read_section_bytes(adapter, strtab)
|
||||||
|
if not strtab_bytes then return nil, "truncated_section_headers" end
|
||||||
|
if s.sh_offset + s.sh_size > file_size then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local symtab_bytes = M.read_section_bytes(adapter, s)
|
||||||
|
if not symtab_bytes then return nil, "truncated_section_headers" end
|
||||||
|
local n = #symtab_bytes / M.ELF32_SYM.sym_entry_bytes
|
||||||
|
for j = 0, n - 1 do
|
||||||
|
local e = s.sh_offset + j * M.ELF32_SYM.sym_entry_bytes
|
||||||
|
local st_name = M.read_u32(adapter, e + M.ELF32_SYM.st_name)
|
||||||
|
if st_name then
|
||||||
|
local st_value = M.read_u32(adapter, e + M.ELF32_SYM.st_value)
|
||||||
|
local st_size = M.read_u32(adapter, e + M.ELF32_SYM.st_size)
|
||||||
|
local st_info = M.read_u8(adapter, e + M.ELF32_SYM.st_info)
|
||||||
|
-- st_shndx is at offset 14 (2 bytes) — derived from the layout
|
||||||
|
-- the metaprogram reads too. Inline the read to keep the
|
||||||
|
-- adapter as the only I/O surface.
|
||||||
|
local b1 = M.read_u8(adapter, e + 14)
|
||||||
|
local b2 = M.read_u8(adapter, e + 15)
|
||||||
|
if not (b1 and b2) then
|
||||||
|
return nil, "truncated_section_headers"
|
||||||
|
end
|
||||||
|
local st_shndx = b1 + b2 * 0x100
|
||||||
|
local name = M.get_str(strtab_bytes, st_name) or ""
|
||||||
|
if name ~= "" then
|
||||||
|
symbols[name] = {
|
||||||
|
value = st_value,
|
||||||
|
size = st_size,
|
||||||
|
info = st_info,
|
||||||
|
shndx = st_shndx,
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return symbols, nil
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
+209
-212
@@ -1,13 +1,8 @@
|
|||||||
--- elf_dwarf.lua — ELF32 + DWARF + atoms source-map utilities.
|
--- elf_dwarf.lua — ELF32 + DWARF + atoms source-map utilities.
|
||||||
--- All ELF32 + DWARF-specific code lives here.
|
|
||||||
---
|
|
||||||
--- **What this module contains:**
|
--- **What this module contains:**
|
||||||
--- - **Format-constant tables** (the byte-offset / opcode / size encyclopedias for ELF32, DWARF4 aranges, DWARF5 rnglists, DWARF line-program, MIPS).
|
--- - **Format-constant tables** (the byte-offset / opcode / size encyclopedias for ELF32, DWARF4 aranges, DWARF5 rnglists, DWARF line-program, MIPS).
|
||||||
--- Every constant carries a spec:` comment naming the spec section that defines it.
|
--- Every constant carries a spec:` comment naming the spec section that defines it.
|
||||||
--- - **I/O helpers**: little-endian byte read/write, ELF32 section walker, nm symbol reader, source-map parser, native directory glob.
|
--- - **I/O helpers**: little-endian byte read/write, ELF32 section walker, nm symbol reader, source-map parser, native directory glob.
|
||||||
---
|
|
||||||
--- **Conventions:** tabs (1/level), EmmyLua annotations, no regex,
|
|
||||||
--- Lua 5.3 compatible.
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Native dependencies
|
-- Native dependencies
|
||||||
@@ -16,6 +11,11 @@
|
|||||||
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
|
-- lfs is wired into package.cpath by `duffle_paths.lua` (vendored under `toolchain/lfs/lfs.dll`).
|
||||||
local lfs = require("lfs")
|
local lfs = require("lfs")
|
||||||
|
|
||||||
|
-- scripts/elf32.lua contains format-constant tables + the byte-level walker.
|
||||||
|
-- The this file re-exports `read_u32_le` / `read_u16_le` (and the DWARF32 terminator).
|
||||||
|
-- TODO(Ed): Remove re-export.
|
||||||
|
local E = require("elf32")
|
||||||
|
|
||||||
local M = {}
|
local M = {}
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -60,19 +60,19 @@ M.DW_AT = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
M.DW_FORM = {
|
M.DW_FORM = {
|
||||||
addr = 0x01,
|
addr = 0x01,
|
||||||
data1 = 0x0B,
|
data1 = 0x0B,
|
||||||
data2 = 0x05,
|
data2 = 0x05,
|
||||||
data4 = 0x06,
|
data4 = 0x06,
|
||||||
string = 0x08,
|
string = 0x08,
|
||||||
strp = 0x0E,
|
strp = 0x0E,
|
||||||
exprloc = 0x18,
|
exprloc = 0x18,
|
||||||
ref4 = 0x13,
|
ref4 = 0x13,
|
||||||
udata = 0x0F,
|
udata = 0x0F,
|
||||||
ref_sig8 = 0x20,
|
ref_sig8 = 0x20,
|
||||||
implicit_const = 0x21,
|
implicit_const = 0x21,
|
||||||
flag_present = 0x19,
|
flag_present = 0x19,
|
||||||
sec_offset = 0x17,
|
sec_offset = 0x17,
|
||||||
}
|
}
|
||||||
|
|
||||||
M.DW_ATE = {
|
M.DW_ATE = {
|
||||||
@@ -104,34 +104,16 @@ M.MIPS_BYTES_PER_WORD = 0x04
|
|||||||
-- ----------------------------------------------------------------------------
|
-- ----------------------------------------------------------------------------
|
||||||
-- ELF32 (System V ABI gABI v1.2)
|
-- ELF32 (System V ABI gABI v1.2)
|
||||||
-- ----------------------------------------------------------------------------
|
-- ----------------------------------------------------------------------------
|
||||||
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors,
|
--- **Wire-offset contract:** format offsets, fixed-width reader offsets, LEB/parser cursors, and section-relative values are zero-based wire offsets.
|
||||||
--- and section-relative values are zero-based wire offsets. Only Lua string APIs receive
|
--- Only Lua string APIs receive a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
|
||||||
--- a `+ 1` conversion at their boundary (`byte`, `sub`, and `find`).
|
--- ELF/DWARF field offsets are expressed in hex so they map directly to the zero-based byte positions in the binary file.
|
||||||
---
|
---
|
||||||
--- ELF/DWARF field offsets are expressed in hex so they map directly to the
|
--- The ELF32 header / section / sym layout tables are within scripts/elf32.lua.
|
||||||
--- zero-based byte positions in the binary file.
|
--- The metaprogram re-exports the DWARF32 initial-length terminator.
|
||||||
|
|
||||||
|
--- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
||||||
--- spec: System V ABI gABI v1.2 §"ELF Header" (Table 1) + §"Section Header Table"
|
M.dw_dwarf32_terminator = E.dw_dwarf32_terminator
|
||||||
M.ELF32 = {
|
-- TODO(Ed): Remove re-export.
|
||||||
magic_offset = 0x00, -- 4-byte magic "\127ELF" at file offset 0x00
|
|
||||||
magic = "\127ELF",
|
|
||||||
class_offset = 0x04, -- 1-byte; 1 = ELF32, 2 = ELF64
|
|
||||||
class_elf32 = 1,
|
|
||||||
endian_offset = 0x05, -- 1-byte; 1 = little-endian, 2 = big-endian
|
|
||||||
endian_little = 1,
|
|
||||||
header_bytes = 0x34, -- spec: gABI v1.2 §"ELF Header" — ELF32 header is 52 bytes total
|
|
||||||
e_shoff_offset = 0x20, -- 4-byte LE; section-header table file offset
|
|
||||||
e_shentsize_offset = 0x2E, -- 2-byte LE; section-header entry size in bytes
|
|
||||||
e_shnum_offset = 0x30, -- 2-byte LE; number of section headers
|
|
||||||
e_shstrndx_offset = 0x32, -- 2-byte LE; index of section-name string table
|
|
||||||
sh_size_bytes = 0x28, -- spec: gABI v1.2 §"Section Header Table" — each entry is 40 bytes
|
|
||||||
sh_name_offset = 0x00, -- 4-byte LE; offset into .shstrtab
|
|
||||||
sh_type_offset = 0x04, -- 4-byte LE; section type (SHT_*)
|
|
||||||
sh_offset_offset = 0x10, -- 4-byte LE; section's file offset
|
|
||||||
sh_size_offset = 0x14, -- 4-byte LE; section's size in bytes
|
|
||||||
dw_dwarf32_terminator = 0xFFFFFFFF, -- spec: DWARF4 spec §7.4 — 32-bit DWARF initial-length terminator
|
|
||||||
}
|
|
||||||
|
|
||||||
-- ----------------------------------------------------------------------------
|
-- ----------------------------------------------------------------------------
|
||||||
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
|
-- DWARF4 .debug_aranges (per DWARF5 spec §7.4 — Address Range Table)
|
||||||
@@ -192,8 +174,7 @@ M.DWARF_LINE_OPS = {
|
|||||||
DW_LNE_end_sequence = 1, -- spec: §6.2.5.3
|
DW_LNE_end_sequence = 1, -- spec: §6.2.5.3
|
||||||
DW_LNE_set_address = 2, -- spec: §6.2.5.3
|
DW_LNE_set_address = 2, -- spec: §6.2.5.3
|
||||||
-- Standard opcode header (§6.2.5.1)
|
-- Standard opcode header (§6.2.5.1)
|
||||||
-- opcode_base + line_range are 1-byte header fields; hex so they map
|
-- opcode_base + line_range are 1-byte header fields; hex so they map directly to the line-program header byte sequence.
|
||||||
-- directly to the line-program header byte sequence.
|
|
||||||
-- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics.
|
-- line_base stays signed decimal (=-5) since 0xFB obscures the spec semantics.
|
||||||
opcode_base = 0x0D,
|
opcode_base = 0x0D,
|
||||||
line_base = -5,
|
line_base = -5,
|
||||||
@@ -249,41 +230,35 @@ M.DWARF5_DEBUG_LINE = {
|
|||||||
--- Read a 4-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
--- Read a 4-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||||
--- Equivalent to `string.unpack("<I4", buf, off + 1)` but avoids the table-return shape + works under LuaJIT 2.1
|
--- Equivalent to `string.unpack("<I4", buf, off + 1)` but avoids the table-return shape + works under LuaJIT 2.1
|
||||||
--- (which has partial `string.unpack` coverage).
|
--- (which has partial `string.unpack` coverage).
|
||||||
---
|
|
||||||
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
|
--- **Convention:** `off` is a zero-based wire offset; `+ 1` is applied only at the `string.byte` boundary.
|
||||||
---
|
---
|
||||||
--- **Byte weights** are written as `0x100`, `0x10000`, `0x1000000` (i.e. 2^8, 2^16, 2^24) so the LE byte positions are visually explicit:
|
--- Thin forwarder: the canonical implementation lives in scripts/elf32.lua.
|
||||||
--- byte 0 contributes its value directly; byte 1 is shifted left by 8
|
--- The "second caller lifts" pattern keeps the metaprogram side fluent
|
||||||
--- (= 0x100); byte 2 by 16 (= 0x10000); byte 3 by 24 (= 0x1000000).
|
--- (`M.read_u32_le(buf, off)`) while the body is deduped.
|
||||||
--- @param buf string
|
--- @param buf string
|
||||||
--- @param off integer -- zero-based wire offset
|
--- @param off integer -- zero-based wire offset
|
||||||
--- @return integer
|
--- @return integer
|
||||||
function M.read_u32_le(buf, off)
|
function M.read_u32_le(buf, off)
|
||||||
local byte_off = off + 1
|
return E.read_u32_le(buf, off)
|
||||||
return buf:byte(byte_off)
|
|
||||||
+ buf:byte(byte_off + 0x01) * 0x00000100
|
|
||||||
+ buf:byte(byte_off + 0x02) * 0x00010000
|
|
||||||
+ buf:byte(byte_off + 0x03) * 0x01000000
|
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
--- Read a 2-byte little-endian unsigned integer from `buf` at zero-based wire offset `off`.
|
||||||
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
|
--- (`off` is zero-based; `+ 1` is applied only at the `string.byte` boundary.)
|
||||||
|
--- Thin forwarder — see `M.read_u32_le` for the rationale.
|
||||||
--- @param buf string
|
--- @param buf string
|
||||||
--- @param off integer -- zero-based wire offset
|
--- @param off integer -- zero-based wire offset
|
||||||
--- @return integer
|
--- @return integer
|
||||||
function M.read_u16_le(buf, off)
|
function M.read_u16_le(buf, off)
|
||||||
local byte_off = off + 1
|
return E.read_u16_le(buf, off)
|
||||||
return buf:byte(byte_off) + buf:byte(byte_off + 0x01) * 0x00000100
|
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
|
-- Pure-Lua 5.3 LEB128 readers (no `bit` library). `2^shift` arithmetic matches the existing parser.
|
||||||
-- Offsets are 0-based; returns (value, next_pos).
|
-- Offsets are 0-based; returns (value, next_pos).
|
||||||
-- Promoted from `local function` to M.* exports so passes/dwarf_injection.lua
|
-- Promoted from `local function` to M.* exports so passes/dwarf_injection.lua can import them as file-scope locals per the 2nd-caller lift precedent
|
||||||
-- can import them as file-scope locals per the 2nd-caller lift precedent
|
|
||||||
-- (the uleb128 + sleb128 encoders were promoted the same way).
|
-- (the uleb128 + sleb128 encoders were promoted the same way).
|
||||||
function M.read_uleb128_at(buf, pos)
|
function M.read_uleb128_at(buf, pos)
|
||||||
local value, shift = 0, 0
|
local value, shift = 0, 0
|
||||||
local len = #buf
|
local len = #buf
|
||||||
while pos < len do
|
while pos < len do
|
||||||
local b = buf:byte(pos + 1)
|
local b = buf:byte(pos + 1)
|
||||||
value = value + (b % 0x80) * (2 ^ shift)
|
value = value + (b % 0x80) * (2 ^ shift)
|
||||||
@@ -427,13 +402,13 @@ local function read_form_value(buf, str_buf, pos, form)
|
|||||||
-- The constant is declared in the abbrev; no value bytes in the DIE.
|
-- The constant is declared in the abbrev; no value bytes in the DIE.
|
||||||
return nil, pos
|
return nil, pos
|
||||||
elseif form == M.DW_FORM.ref_sig8 then
|
elseif form == M.DW_FORM.ref_sig8 then
|
||||||
-- DW_FORM_ref_sig8 (DWARF5 §7.4.2): an 8-byte value identifying a type
|
-- DW_FORM_ref_sig8 (DWARF5 §7.4.2): An 8-byte value identifying a type by signature.
|
||||||
-- by signature. The low 4 bytes (LE) are the type signature (content hash);
|
-- The low 4 bytes (LE) are the type signature (content hash);
|
||||||
-- the high 4 bytes (LE) are a CU-relative offset into the matching type unit.
|
-- The high 4 bytes (LE) are a CU-relative offset into the matching type unit.
|
||||||
-- Consumers use the low 4 to look up the type unit (see M.find_type_unit_by_signature)
|
-- Consumers use the low 4 to look up the type unit (see M.find_type_unit_by_signature)
|
||||||
-- then the high 4 to resolve the specific type within it.
|
-- then the high 4 to resolve the specific type within it.
|
||||||
-- Return the low 4 as the primary value to preserve the (value, next_pos) shape;
|
-- Return the low 4 as the primary value to preserve the (value, next_pos) shape;
|
||||||
-- the high 4 is exposed via M.read_ref_sig8 (which returns both halves).
|
-- the high 4 is exposed via M.read_ref_sig8 (which returns both halves).
|
||||||
local _, _, next_pos = M.read_ref_sig8(buf, pos)
|
local _, _, next_pos = M.read_ref_sig8(buf, pos)
|
||||||
return M.read_u32_le(buf, pos), next_pos
|
return M.read_u32_le(buf, pos), next_pos
|
||||||
else
|
else
|
||||||
@@ -455,26 +430,26 @@ function M.read_ref_sig8(buf, pos)
|
|||||||
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
|
return M.read_u32_le(buf, pos), M.read_u32_le(buf, pos + 4), pos + 8
|
||||||
end
|
end
|
||||||
|
|
||||||
-- DWARF5 §7.5.6 (Type Entries).
|
--- DWARF5 §7.5.6 (Type Entries).
|
||||||
-- Walk all units in `info` and return the 0-based offset of the first unit
|
--- Walk all units in `info` and return the 0-based offset of the first unit whose `DW_AT_type_signature`
|
||||||
-- whose `DW_AT_type_signature` (8-byte value at the end of the unit header) equals `target_sig`.
|
--- (8-byte value at the end of the unit header) equals `target_sig`.
|
||||||
-- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
--- The signature is interpreted as two 32-bit halves (low/high) per the read_ref_sig8 contract;
|
||||||
-- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
--- we match both halves (i.e. the 8-byte value as a whole). Returns nil if no matching unit exists.
|
||||||
--
|
---
|
||||||
-- Unit header layout (from pos 0):
|
--- Unit header layout (from pos 0):
|
||||||
-- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
--- unit_length(4) + version(2) + unit_type(1) + address_size(1) + debug_abbrev_offset(4)
|
||||||
-- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
--- followed by type_unit_specific fields: type_signature(8) + type_offset(4)
|
||||||
-- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
--- The type_signature is at byte offset 8 of the body (right after debug_abbrev_offset).
|
||||||
-- @param info string -- the .debug_info section bytes
|
--- @param info string -- the .debug_info section bytes
|
||||||
-- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
--- @param target_sig_lo integer -- low 4 bytes (LE) of the desired signature
|
||||||
-- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
--- @param target_sig_hi integer -- high 4 bytes (LE) of the desired signature
|
||||||
-- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
--- @return integer|nil, integer|nil -- unit offset, type_offset within the unit
|
||||||
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
||||||
local pos = 0
|
local pos = 0
|
||||||
local section_len = #info
|
local section_len = #info
|
||||||
while pos + 4 < section_len do
|
while pos + 4 < section_len do
|
||||||
local unit_length = M.read_u32_le(info, pos)
|
local unit_length = M.read_u32_le(info, pos)
|
||||||
if unit_length == 0xFFFFFFFF then
|
if unit_length == 0xFFFFFFFF then
|
||||||
return nil, nil -- DWARF64 not supported
|
return nil, nil -- DWARF64 not supported
|
||||||
end
|
end
|
||||||
-- unit_length is the body size, NOT including the 4-byte unit_length field itself.
|
-- unit_length is the body size, NOT including the 4-byte unit_length field itself.
|
||||||
@@ -503,9 +478,9 @@ function M.find_type_unit_by_signature(info, target_sig_lo, target_sig_hi)
|
|||||||
-- byte 8-15: type_signature (8)
|
-- byte 8-15: type_signature (8)
|
||||||
-- byte 16-19: type_offset (4)
|
-- byte 16-19: type_offset (4)
|
||||||
local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed
|
local unit_type = info:byte(body_start + 2 + 1) -- 0-based +2 = unit_type in 1-indexed
|
||||||
if unit_type == 0x02 then -- DW_UT_type
|
if unit_type == 0x02 then -- DW_UT_type
|
||||||
local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed
|
local sig_lo, sig_hi, _ = M.read_ref_sig8(info, body_start + 8) -- 0-based +8 = type_signature in 1-indexed
|
||||||
if sig_lo == target_sig_lo and sig_hi == target_sig_hi then
|
if sig_lo == target_sig_lo and sig_hi == target_sig_hi then
|
||||||
local type_offset = M.read_u32_le(info, body_start + 16) -- 0-based +16 = type_offset in 1-indexed
|
local type_offset = M.read_u32_le(info, body_start + 16) -- 0-based +16 = type_offset in 1-indexed
|
||||||
return pos, type_offset
|
return pos, type_offset
|
||||||
end
|
end
|
||||||
@@ -552,7 +527,7 @@ end
|
|||||||
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
|
--- (we walk all `e_shnum` headers regardless of how many names are requested, to find the .shstrtab first).
|
||||||
--- For frequent callers, pass the union of all needed sections in one call.
|
--- For frequent callers, pass the union of all needed sections in one call.
|
||||||
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
|
-- Can add `.debug_info` + `.debug_loc` + `.debug_str_offsets` to the list without writing a 2nd ELF walker.
|
||||||
--- @param elf_path Path
|
--- @param elf_path Path
|
||||||
--- @param section_names string[] -- list of section names to read
|
--- @param section_names string[] -- list of section names to read
|
||||||
--- @return table<string, string>
|
--- @return table<string, string>
|
||||||
function M.read_elf_sections(elf_path, section_names)
|
function M.read_elf_sections(elf_path, section_names)
|
||||||
@@ -571,75 +546,64 @@ function M.read_elf_sections(elf_path, section_names)
|
|||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
local f = io.open(elf_path, "rb")
|
local f = io.open(elf_path, "rb")
|
||||||
if not f then
|
if not f then
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] io.open failed: %s\n", elf_path))
|
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] io.open failed: %s\n", elf_path))
|
||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Read the ELF32 header.
|
local file_size
|
||||||
local header = f:read(M.ELF32.header_bytes)
|
do
|
||||||
if not header or #header < M.ELF32.header_bytes then
|
f:seek("end", 0)
|
||||||
io.stderr:write("[elf_dwarf.read_elf_sections] ELF too small for ELF32 header\n")
|
file_size = f:seek("cur", 0)
|
||||||
|
end
|
||||||
|
local adapter = {
|
||||||
|
read_u8_at = function(offset)
|
||||||
|
f:seek("set", offset)
|
||||||
|
local b = f:read(1)
|
||||||
|
if not b then return nil end
|
||||||
|
return b:byte()
|
||||||
|
end,
|
||||||
|
read_u16_at = function(offset)
|
||||||
|
f:seek("set", offset)
|
||||||
|
local b1 = f:read(1)
|
||||||
|
local b2 = f:read(1)
|
||||||
|
if not b1 or not b2 then return nil end
|
||||||
|
return b1:byte() + b2:byte() * 0x100
|
||||||
|
end,
|
||||||
|
read_u32_at = function(offset)
|
||||||
|
f:seek("set", offset)
|
||||||
|
local b1 = f:read(1)
|
||||||
|
local b2 = f:read(1)
|
||||||
|
local b3 = f:read(1)
|
||||||
|
local b4 = f:read(1)
|
||||||
|
if not b1 or not b2 or not b3 or not b4 then return nil end
|
||||||
|
return b1:byte() + b2:byte() * 0x100
|
||||||
|
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
|
||||||
|
end,
|
||||||
|
read_size = function() return file_size end,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Delegate the header parse + section walk to E.*.
|
||||||
|
local hdr, hdr_err = E.parse_elf32_headers(adapter)
|
||||||
|
if not hdr then
|
||||||
|
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] header parse failed: %s\n", tostring(hdr_err)))
|
||||||
f:close()
|
f:close()
|
||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Sanity-check magic + class + endianness.
|
local sections, walk_err = E.walk_sections(adapter, hdr)
|
||||||
if header:sub(M.ELF32.magic_offset + 1, M.ELF32.magic_offset + 0x04) ~= M.ELF32.magic then
|
if not sections then
|
||||||
io.stderr:write("[elf_dwarf.read_elf_sections] not an ELF file\n")
|
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] section walk failed: %s\n", tostring(walk_err)))
|
||||||
f:close()
|
|
||||||
return result
|
|
||||||
end
|
|
||||||
if header:byte(M.ELF32.class_offset + 1) ~= M.ELF32.class_elf32 then
|
|
||||||
io.stderr:write(string.format("[elf_dwarf.read_elf_sections] not ELF32 (class=%d)\n", header:byte(M.ELF32.class_offset + 1)))
|
|
||||||
f:close()
|
|
||||||
return result
|
|
||||||
end
|
|
||||||
if header:byte(M.ELF32.endian_offset + 1) ~= M.ELF32.endian_little then
|
|
||||||
io.stderr:write("[elf_dwarf.read_elf_sections] not little-endian; unsupported\n")
|
|
||||||
f:close()
|
f:close()
|
||||||
return result
|
return result
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Parse section-header table location + dimensions from the header.
|
-- Resolve the requested sections.
|
||||||
local e_shoff = M.read_u32_le(header, M.ELF32.e_shoff_offset)
|
for _, s in ipairs(sections) do
|
||||||
local e_shentsize = M.read_u16_le(header, M.ELF32.e_shentsize_offset)
|
if wanted[s.name] then
|
||||||
local e_shnum = M.read_u16_le(header, M.ELF32.e_shnum_offset)
|
local bytes = E.read_section_bytes(adapter, s)
|
||||||
local e_shstrndx = M.read_u16_le(header, M.ELF32.e_shstrndx_offset)
|
if bytes then result[s.name] = bytes end
|
||||||
|
|
||||||
-- Read the section-header string table (.shstrtab) so we can resolve section names from their `sh_name` offsets.
|
|
||||||
f:seek("set", e_shoff + e_shstrndx * e_shentsize)
|
|
||||||
local strtab_hdr = f:read(e_shentsize)
|
|
||||||
if not strtab_hdr or #strtab_hdr < e_shentsize then
|
|
||||||
io.stderr:write("[elf_dwarf.read_elf_sections] could not read .shstrtab header\n")
|
|
||||||
f:close()
|
|
||||||
return result
|
|
||||||
end
|
|
||||||
local strtab_offset = M.read_u32_le(strtab_hdr, M.ELF32.sh_offset_offset)
|
|
||||||
local strtab_size = M.read_u32_le(strtab_hdr, M.ELF32.sh_size_offset)
|
|
||||||
f:seek("set", strtab_offset)
|
|
||||||
local strtab = f:read(strtab_size) or ""
|
|
||||||
|
|
||||||
-- Walk all section headers; collect (offset, size) for the wanted names.
|
|
||||||
local function read_section_bytes(sh_offset, sh_size)
|
|
||||||
f:seek("set", sh_offset)
|
|
||||||
return f:read(sh_size) or ""
|
|
||||||
end
|
|
||||||
|
|
||||||
for sh_idx = 0, e_shnum - 1 do
|
|
||||||
f:seek("set", e_shoff + sh_idx * e_shentsize)
|
|
||||||
local sh = f:read(e_shentsize)
|
|
||||||
if not sh or #sh < e_shentsize then break end
|
|
||||||
local sh_name = M.read_u32_le(sh, M.ELF32.sh_name_offset)
|
|
||||||
local sh_offset = M.read_u32_le(sh, M.ELF32.sh_offset_offset)
|
|
||||||
local sh_size = M.read_u32_le(sh, M.ELF32.sh_size_offset)
|
|
||||||
|
|
||||||
-- Extract the name (null-terminated C string in strtab).
|
|
||||||
local name_end = strtab:find("\0", sh_name + 1, true) or (sh_name + 1)
|
|
||||||
local name = strtab:sub(sh_name + 1, name_end - 1)
|
|
||||||
if wanted[name] then
|
|
||||||
result[name] = read_section_bytes(sh_offset, sh_size)
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -656,49 +620,87 @@ end
|
|||||||
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
--- - We filter on STB_GLOBAL (high nibble of st_info = 1) to match `nm`'s default (external symbols only). STB_WEAK excluded.
|
||||||
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
--- - The `code_` prefix is stripped (MipsAtom_ macros emit bare atom names, no `code_` prefix).
|
||||||
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
--- - `st_size > 0` filter excludes undefined/imported symbols.
|
||||||
|
---
|
||||||
--- @param elf_path Path
|
--- @param elf_path Path
|
||||||
--- @return table<string, {integer, integer}>
|
--- @return table<string, {integer, integer}>
|
||||||
function M.read_nm(elf_path)
|
function M.read_nm(elf_path)
|
||||||
local addrs = {}
|
local addrs = {}
|
||||||
|
|
||||||
-- Read .symtab + .strtab via the existing ELF walker (no subprocess).
|
-- Existence check first; an empty or missing ELF returns an empty map.
|
||||||
local sections = M.read_elf_sections(elf_path, {".symtab", ".strtab"})
|
if lfs.attributes(elf_path, "mode") ~= "file" then
|
||||||
local symtab = sections[".symtab"]
|
|
||||||
local strtab = sections[".strtab"]
|
|
||||||
if not symtab or not strtab or #symtab == 0 or #strtab == 0 then
|
|
||||||
-- No symbol table (e.g. stripped ELF). Return empty.
|
|
||||||
return addrs
|
return addrs
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Iterate the 16-byte ELF32 symtab entries.
|
local f = io.open(elf_path, "rb")
|
||||||
-- Each entry (zero-based): st_name at 0, st_value at 4, st_size at 8, st_info at 12, st_other at 13, st_shndx at 14.
|
if not f then
|
||||||
local SYM_ENTRY_BYTES = 0x10
|
return addrs
|
||||||
local SYM_ST_NAME = 0x00
|
end
|
||||||
local SYM_ST_VALUE = 0x04
|
|
||||||
local SYM_ST_SIZE = 0x08
|
-- Build the file adapter for E.*.
|
||||||
local SYM_ST_INFO = 0x0C
|
local file_size
|
||||||
local n_syms = #symtab / SYM_ENTRY_BYTES
|
do
|
||||||
for i = 0, n_syms - 1 do
|
f:seek("end", 0)
|
||||||
local entry_off = i * SYM_ENTRY_BYTES
|
file_size = f:seek("cur", 0)
|
||||||
local st_info = symtab:byte(entry_off + SYM_ST_INFO + 1)
|
end
|
||||||
-- High nibble = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
local adapter = {
|
||||||
-- Use math.floor(/16) instead of bit.rshift for LuaJIT 2.1 compat
|
read_u8_at = function(offset)
|
||||||
-- (LuaJIT's `>>` is 5.3+, but math.floor(x/16) works on all versions).
|
f:seek("set", offset)
|
||||||
local binding = math.floor(st_info / 16)
|
local b = f:read(1)
|
||||||
if binding == 0 or binding == 1 then -- STB_LOCAL or STB_GLOBAL
|
if not b then return nil end
|
||||||
local st_size = M.read_u32_le(symtab, entry_off + SYM_ST_SIZE)
|
return b:byte()
|
||||||
if st_size > 0 then
|
end,
|
||||||
local st_name_off = M.read_u32_le(symtab, entry_off + SYM_ST_NAME)
|
read_u16_at = function(offset)
|
||||||
-- Extract the name from .strtab (null-terminated C string).
|
f:seek("set", offset)
|
||||||
local name_end = strtab:find("\0", st_name_off + 1, true) or (st_name_off + 1)
|
local b1 = f:read(1)
|
||||||
local name = strtab:sub(st_name_off + 1, name_end - 1)
|
local b2 = f:read(1)
|
||||||
-- Filter: keep all symbol-table symbols (atoms emit their name as the bare `<name>` — MipsAtom_ macros strip the `code_` prefix).
|
if not b1 or not b2 then return nil end
|
||||||
-- The atoms_source_map pass already filters out non-atom symbols via the source-map.txt cross-ref.
|
return b1:byte() + b2:byte() * 0x100
|
||||||
if name and #name > 0 then
|
end,
|
||||||
local st_value = M.read_u32_le(symtab, entry_off + SYM_ST_VALUE)
|
read_u32_at = function(offset)
|
||||||
addrs[name] = { st_value, st_size }
|
f:seek("set", offset)
|
||||||
end
|
local b1 = f:read(1)
|
||||||
end
|
local b2 = f:read(1)
|
||||||
|
local b3 = f:read(1)
|
||||||
|
local b4 = f:read(1)
|
||||||
|
if not b1 or not b2 or not b3 or not b4 then return nil end
|
||||||
|
return b1:byte() + b2:byte() * 0x100
|
||||||
|
+ b3:byte() * 0x10000 + b4:byte() * 0x1000000
|
||||||
|
end,
|
||||||
|
read_size = function() return file_size end,
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Delegate the header + section walk to E.*.
|
||||||
|
local hdr, hdr_err = E.parse_elf32_headers(adapter)
|
||||||
|
if not hdr then
|
||||||
|
io.stderr:write(string.format("[elf_dwarf.read_nm] header parse failed: %s\n", tostring(hdr_err)))
|
||||||
|
f:close()
|
||||||
|
return addrs
|
||||||
|
end
|
||||||
|
|
||||||
|
local sections, walk_err = E.walk_sections(adapter, hdr)
|
||||||
|
if not sections then
|
||||||
|
io.stderr:write(string.format("[elf_dwarf.read_nm] section walk failed: %s\n", tostring(walk_err)))
|
||||||
|
f:close()
|
||||||
|
return addrs
|
||||||
|
end
|
||||||
|
|
||||||
|
-- E.collect_symbols returns every defined symbol (no binding filter).
|
||||||
|
-- The metaprogram then applies its STB_LOCAL / STB_GLOBAL + size>0 filter, matching `nm`'s default (external symbols only).
|
||||||
|
local symbols, sym_err = E.collect_symbols(adapter, sections)
|
||||||
|
if not symbols then
|
||||||
|
io.stderr:write(string.format("[elf_dwarf.read_nm] symbol collection failed: %s\n", tostring(sym_err)))
|
||||||
|
f:close()
|
||||||
|
return addrs
|
||||||
|
end
|
||||||
|
|
||||||
|
f:close()
|
||||||
|
|
||||||
|
for name, entry in pairs(symbols) do
|
||||||
|
-- High nibble of st_info = binding (STB_LOCAL=0, STB_GLOBAL=1, STB_WEAK=2).
|
||||||
|
-- math.floor(/16) is portable across LuaJIT 2.0/2.1 and plain Lua 5.x.
|
||||||
|
local binding = math.floor(entry.info / 16)
|
||||||
|
if (binding == 0 or binding == 1) and entry.size > 0 then
|
||||||
|
addrs[name] = { entry.value, entry.size }
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -779,8 +781,8 @@ function M.sleb128(n)
|
|||||||
local b = n % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
local b = n % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
||||||
n = (n - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
n = (n - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
||||||
-- Termination: remaining value bits fit in the sign bit of the last byte.
|
-- Termination: remaining value bits fit in the sign bit of the last byte.
|
||||||
if n == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
|
if n == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
|
||||||
if n == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
|
if n == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
|
||||||
if more then b = b + LEB_CONT_BIT end
|
if more then b = b + LEB_CONT_BIT end
|
||||||
bytes[#bytes + 1] = string.char(b)
|
bytes[#bytes + 1] = string.char(b)
|
||||||
end
|
end
|
||||||
@@ -811,14 +813,14 @@ end
|
|||||||
--- @param n integer -- any integer (negative allowed)
|
--- @param n integer -- any integer (negative allowed)
|
||||||
--- @return integer
|
--- @return integer
|
||||||
function M.sleb128_size(n)
|
function M.sleb128_size(n)
|
||||||
local more = true
|
local more = true
|
||||||
local bytes = 0
|
local bytes = 0
|
||||||
local v = n
|
local v = n
|
||||||
while more do
|
while more do
|
||||||
local b = v % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
local b = v % (LEB_DATA_MASK + 1) -- extract low 7 bits
|
||||||
v = (v - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
v = (v - b) / (LEB_DATA_MASK + 1) -- arithmetic shift right by 7
|
||||||
if v == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
|
if v == 0 and b < SLEB_SIGN_BIT then more = false end -- positive terminator
|
||||||
if v == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
|
if v == -1 and b >= SLEB_SIGN_BIT then more = false end -- negative terminator
|
||||||
if more then b = b + LEB_CONT_BIT end
|
if more then b = b + LEB_CONT_BIT end
|
||||||
bytes = bytes + 1
|
bytes = bytes + 1
|
||||||
end
|
end
|
||||||
@@ -836,12 +838,11 @@ end
|
|||||||
--- * The `.debug_line` section may contain MULTIPLE line-program units
|
--- * The `.debug_line` section may contain MULTIPLE line-program units
|
||||||
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
|
--- File indices are 1-based, **per unit**; we concatenate all units and the index ranges from 1..N₁ in unit 1, N₁+1..N₁+N₂ in unit 2, etc.
|
||||||
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
|
--- Per-unit indices (the way gcc emits them, and the way `DW_LNS_set_file` references them in the line program)
|
||||||
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom
|
--- are returned via the `basename_to_index` map only when the unit boundary happens to align with the metaprogram's per-atom `inv.call_file`
|
||||||
--- `inv.call_file` (true today for hello_joypad — the C unit is the LAST unit, and atom-side file indices fit 1-based).
|
|
||||||
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
|
--- * Per spec, the `.debug_line_str` section (DWARF5 §7.5.6) holds the strings referenced by `DW_FORM_line_strp`.
|
||||||
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
|
--- The legacy DWARF3 format embeds strings directly with null terminators. This helper handles BOTH.
|
||||||
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index`
|
--- * File entries may have multiple forms (gcc -gdwarf-5 with `DW_LNCT_directory_index` emits 2 forms: path + dir_index).
|
||||||
--- emits 2 forms: path + dir_index). The helper supports:
|
--- The helper supports:
|
||||||
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
|
--- - DW_FORM_line_strp (DWARF5; offset into .debug_line_str)
|
||||||
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
|
--- - DW_FORM_string (DWARF4-compat; inline null-terminated in .debug_line)
|
||||||
--- - DW_FORM_udata (ULEB128)
|
--- - DW_FORM_udata (ULEB128)
|
||||||
@@ -851,51 +852,47 @@ end
|
|||||||
---
|
---
|
||||||
--- Behavior on failure: writes to stderr and returns nil.
|
--- Behavior on failure: writes to stderr and returns nil.
|
||||||
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
|
--- Helpers consumed by `passes/dwarf_injection.lua::init_file_index_lookup(elf_path)` calls this once at pass start to populate the module-level `basename_to_index` map;
|
||||||
--- downstream `resolve_provenance_file_index(path)` consumers
|
--- downstream `resolve_provenance_file_index(path)` consumers consult the map directly.
|
||||||
--- (which replaced the former hardcoded `ATOM_SOURCE_FILE_INDEX` + `PROVENANCE_BASENAME_TO_FILE_INDEX` table per `conductor/tracks/dwarf_file_index_lookup_20260731/`)
|
|
||||||
--- consult the map directly.
|
|
||||||
---
|
---
|
||||||
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice;
|
--- @param elf_path string -- absolute path to the post-link ELF (typically the gcc-emitted `.elf` BEFORE dwarf_injector's splice; both shapes work since the splice preserves `.debug_line`)
|
||||||
--- both shapes work since the splice preserves `.debug_line`)
|
|
||||||
--- @return table|nil, table|nil, table|nil
|
--- @return table|nil, table|nil, table|nil
|
||||||
--- basename_to_index: { [basename] = 1-based-per-unit-file-index, ... }
|
--- basename_to_index: { [basename] = 1-based-per-unit-file-index, ... }
|
||||||
--- basenames: { [1-based-per-unit-file-index] = basename, ... }
|
--- basenames: { [1-based-per-unit-file-index] = basename, ... }
|
||||||
--- paths: { [1-based-per-unit-file-index] = full path (mixed slashes), ... }
|
--- paths: { [1-based-per-unit-file-index] = full path (mixed slashes), ... }
|
||||||
function M.read_line_unit_file_table(elf_path)
|
function M.read_line_unit_file_table(elf_path)
|
||||||
local sections = M.read_elf_sections(elf_path, { ".debug_line", ".debug_line_str" })
|
local sections = M.read_elf_sections(elf_path, { ".debug_line", ".debug_line_str" })
|
||||||
local line = sections[".debug_line"]
|
local line = sections[".debug_line"]
|
||||||
local lstr = sections[".debug_line_str"] or ""
|
local lstr = sections[".debug_line_str"] or ""
|
||||||
if not line or line == "" then
|
if not line or line == "" then
|
||||||
io.stderr:write("[elf_dwarf.read_line_unit_file_table] no .debug_line section in: " .. tostring(elf_path) .. "\n")
|
io.stderr:write("[elf_dwarf.read_line_unit_file_table] no .debug_line section in: " .. tostring(elf_path) .. "\n")
|
||||||
return nil
|
return nil
|
||||||
end
|
end
|
||||||
|
|
||||||
local basenames = {}
|
local basenames = {}
|
||||||
local basename_to_index = {}
|
local basename_to_index = {}
|
||||||
local paths = {}
|
local paths = {}
|
||||||
|
|
||||||
--- Read one form-code's bytes from `buf` at position `p` according to `form`.
|
--- Read one form-code's bytes from `buf` at position `p` according to `form`.
|
||||||
--- Returns (value, after) where `value` is:
|
--- Returns (value, after) where `value` is:
|
||||||
--- * the resolved string (DW_FORM_line_strp / DW_FORM_string)
|
--- * the resolved string (DW_FORM_line_strp / DW_FORM_string)
|
||||||
--- * the ULEB128 number (DW_FORM_udata)
|
--- * the ULEB128 number (DW_FORM_udata)
|
||||||
--- * nil + skip-bytes (DW_FORM_data16; we don't surface the MD5)
|
--- * nil + skip-bytes (DW_FORM_data16; we don't surface the MD5)
|
||||||
local function read_form(buf, lstr_buf, p, form)
|
local function read_form(buf, lstr_buf, p, form)
|
||||||
if form == M.DWARF5_DEBUG_LINE.form_line_strp then
|
if form == M.DWARF5_DEBUG_LINE.form_line_strp then
|
||||||
local strp = M.read_u32_le(buf, p)
|
local strp = M.read_u32_le(buf, p)
|
||||||
local end_pos = lstr_buf:find("\0", strp + 1, true) or (#lstr_buf + 1)
|
local end_pos = lstr_buf:find("\0", strp + 1, true) or (#lstr_buf + 1)
|
||||||
return lstr_buf:sub(strp + 1, end_pos - 1), p + M.DWARF5_DEBUG_LINE.form_strp_bytes
|
return lstr_buf:sub(strp + 1, end_pos - 1), p + M.DWARF5_DEBUG_LINE.form_strp_bytes
|
||||||
elseif form == M.DWARF5_DEBUG_LINE.form_string then
|
elseif form == M.DWARF5_DEBUG_LINE.form_string then
|
||||||
local nul = buf:find("\0", p + 1, true) or (#buf + 1)
|
local nul = buf:find("\0", p + 1, true) or (#buf + 1)
|
||||||
return buf:sub(p + 1, nul - 1), nul
|
return buf:sub(p + 1, nul - 1), nul
|
||||||
elseif form == M.DWARF5_DEBUG_LINE.form_udata then
|
elseif form == M.DWARF5_DEBUG_LINE.form_udata then
|
||||||
local v, after = M.read_uleb128_at(buf, p)
|
local v, after = M.read_uleb128_at(buf, p)
|
||||||
return v, after
|
return v, after
|
||||||
elseif form == M.DWARF5_DEBUG_LINE.form_data16 then
|
elseif form == M.DWARF5_DEBUG_LINE.form_data16 then
|
||||||
return nil, p + M.DWARF5_DEBUG_LINE.form_data16_bytes
|
return nil, p + M.DWARF5_DEBUG_LINE.form_data16_bytes
|
||||||
else
|
else
|
||||||
-- Unsupported form in a directory/file-table entry: best-effort skip.
|
-- Unsupported form in a directory/file-table entry: best-effort skip.
|
||||||
-- We do NOT stderr-write because the crt0.s DWARF5 line unit (gcc-as emitted) uses DW_FORM_addr (0x01) for what is effectively a path entry,
|
-- We do NOT stderr-write because the crt0.s DWARF5 line unit (gcc-as emitted) uses DW_FORM_addr (0x01) for what is effectively a path entry, which is non-standard.
|
||||||
-- which is non-standard.
|
|
||||||
-- The C-unit's DWARF3 paths are read via the parallel DWARF3 path and never see this error.
|
-- The C-unit's DWARF3 paths are read via the parallel DWARF3 path and never see this error.
|
||||||
-- Callers should consult `basename_to_index` for the paths they care about and ignore this unit if it produced none.
|
-- Callers should consult `basename_to_index` for the paths they care about and ignore this unit if it produced none.
|
||||||
return nil, p
|
return nil, p
|
||||||
@@ -916,9 +913,9 @@ function M.read_line_unit_file_table(elf_path)
|
|||||||
local dirs = {}
|
local dirs = {}
|
||||||
while up < body_end do
|
while up < body_end do
|
||||||
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
|
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
|
||||||
if nul > body_end then break end
|
if nul > body_end then break end
|
||||||
local len = nul - up - 1
|
local len = nul - up - 1
|
||||||
if len == 0 then up = nul break end
|
if len == 0 then up = nul break end
|
||||||
dirs[#dirs + 1] = buf:sub(up + 1, nul - 1)
|
dirs[#dirs + 1] = buf:sub(up + 1, nul - 1)
|
||||||
up = nul
|
up = nul
|
||||||
end
|
end
|
||||||
@@ -926,14 +923,14 @@ function M.read_line_unit_file_table(elf_path)
|
|||||||
local unit_paths = {}
|
local unit_paths = {}
|
||||||
while up < body_end do
|
while up < body_end do
|
||||||
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
|
local nul = buf:find("\0", up + 1, true) or (body_end + 1)
|
||||||
if nul > body_end or nul == up + 1 then up = nul break end
|
if nul > body_end or nul == up + 1 then up = nul break end
|
||||||
local path = buf:sub(up + 1, nul - 1)
|
local path = buf:sub(up + 1, nul - 1)
|
||||||
up = nul
|
up = nul
|
||||||
local didx, up_next = M.read_uleb128_at(buf, up); up = up_next
|
local didx, up_next = M.read_uleb128_at(buf, up); up = up_next
|
||||||
local _time, up_next2 = M.read_uleb128_at(buf, up); up = up_next2
|
local _time, up_next2 = M.read_uleb128_at(buf, up); up = up_next2
|
||||||
local _size, up_next3 = M.read_uleb128_at(buf, up); up = up_next3
|
local _size, up_next3 = M.read_uleb128_at(buf, up); up = up_next3
|
||||||
local idx = #unit_basenames + 1
|
local idx = #unit_basenames + 1
|
||||||
local bs = path:match("[^/\\]+$") or path
|
local bs = path:match("[^/\\]+$") or path
|
||||||
unit_paths[idx] = path
|
unit_paths[idx] = path
|
||||||
unit_basenames[idx] = bs
|
unit_basenames[idx] = bs
|
||||||
dirs[1] = dirs[1] or "" -- safety: gcc emits "" sentinel dir at 0
|
dirs[1] = dirs[1] or "" -- safety: gcc emits "" sentinel dir at 0
|
||||||
@@ -1006,7 +1003,7 @@ function M.read_line_unit_file_table(elf_path)
|
|||||||
local section_end = #line
|
local section_end = #line
|
||||||
while p + 4 <= section_end do
|
while p + 4 <= section_end do
|
||||||
local unit_length = M.read_u32_le(line, p)
|
local unit_length = M.read_u32_le(line, p)
|
||||||
if unit_length == 0xFFFFFFFF then
|
if unit_length == 0xFFFFFFFF then
|
||||||
io.stderr:write("[elf_dwarf.read_line_unit_file_table] 64-bit DWARF (initial-length 0xFFFFFFFF); not supported\n")
|
io.stderr:write("[elf_dwarf.read_line_unit_file_table] 64-bit DWARF (initial-length 0xFFFFFFFF); not supported\n")
|
||||||
return nil
|
return nil
|
||||||
end
|
end
|
||||||
|
|||||||
@@ -22,11 +22,11 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
--- @field scan table -- Pre-scanned SourceScan payload (from duffle.scan_source)
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field sources SourceFile[]
|
--- @field sources SourceFile[]
|
||||||
@@ -45,28 +45,28 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
|||||||
--- @field warnings table[]
|
--- @field warnings table[]
|
||||||
|
|
||||||
--- @class AtomAnnotation
|
--- @class AtomAnnotation
|
||||||
--- @field line integer -- source line of the atom_info call
|
--- @field line integer -- Source line of the atom_info call
|
||||||
--- @field macro string -- the macro name (always "atom_info" in the new shape)
|
--- @field macro string -- Macro name (always "atom_info" in the new shape)
|
||||||
--- @field name string -- the atom name
|
--- @field name string -- Atom name
|
||||||
--- @field kind string -- always "info"
|
--- @field kind string -- Always "info"
|
||||||
--- @field binds string|nil -- Binds_X name if any
|
--- @field binds string|nil -- Binds_X name if any
|
||||||
--- @field reads string[] -- R_* names (read targets)
|
--- @field reads string[] -- R_* names (read targets)
|
||||||
--- @field writes string[] -- R_* names (write targets)
|
--- @field writes string[] -- R_* names (write targets)
|
||||||
--- @field errors string[]|nil -- parse-time errors from scan_source (atom_info body malformed)
|
--- @field errors string[]|nil -- Parse-time errors from scan_source (atom_info body malformed)
|
||||||
|
|
||||||
--- @class DebugSkipMarker -- sub-shape of scan_source.lua's @class DebugSkipMarker
|
--- @class DebugSkipMarker -- Sub-shape of scan_source.lua's @class DebugSkipMarker
|
||||||
--- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive.
|
--- @field marker_kind string -- Exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive.
|
||||||
--- @field marker_line integer
|
--- @field marker_line integer
|
||||||
--- @field args string|nil -- trimmed text inside the parens (nil when has_parens is false)
|
--- @field args string|nil -- Trimmed text inside the parens (nil when has_parens is false)
|
||||||
--- @field has_parens boolean
|
--- @field has_parens boolean
|
||||||
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
|
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
|
||||||
--- @field pending boolean -- true while awaiting the following declaration
|
--- @field pending boolean -- true while awaiting the following declaration
|
||||||
--- @field superseded_by_marker_line integer|nil -- set on a marker that was bumped out of the pending slot
|
--- @field superseded_by_marker_line integer|nil -- Set on a marker that was bumped out of the pending slot
|
||||||
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
||||||
|
|
||||||
--- @class Finding
|
--- @class Finding
|
||||||
--- @field line integer -- source line (or 0 for pass-level)
|
--- @field line integer -- Source line (or 0 for pass-level)
|
||||||
--- @field msg string -- finding message
|
--- @field msg string -- Finding message
|
||||||
|
|
||||||
--- @class Findings
|
--- @class Findings
|
||||||
--- @field errors Finding[]
|
--- @field errors Finding[]
|
||||||
@@ -74,14 +74,14 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
|||||||
--- @field info Finding[]
|
--- @field info Finding[]
|
||||||
|
|
||||||
--- @class PipeCtx
|
--- @class PipeCtx
|
||||||
--- @field atom_index table<string, AtomAnnotation> -- name -> AtomAnnotation (only kind=="atom")
|
--- @field atom_index table<string, AtomAnnotation> -- Name -> AtomAnnotation (only kind=="atom")
|
||||||
--- @field binds_index table<string, BindsStruct> -- name -> BindsStruct
|
--- @field binds_index table<string, BindsStruct> -- Name -> BindsStruct
|
||||||
--- @field annot_counts table<string, integer> -- name -> annotation count (for unique_annotation check)
|
--- @field annot_counts table<string, integer> -- Name -> annotation count (for unique_annotation check)
|
||||||
--- @field types table<string, RegTypeDefault> -- from scan_source
|
--- @field types table<string, RegTypeDefault> -- From scan_source
|
||||||
--- @field atom_views table<string, AtomViewEntry> -- from scan_source
|
--- @field atom_views table<string, AtomViewEntry> -- From scan_source
|
||||||
--- @field seen_defaults table<string, integer> -- duplicate atom_dbg_reg_default detection
|
--- @field seen_defaults table<string, integer> -- Duplicate atom_dbg_reg_default detection
|
||||||
--- @field seen_field table<string, integer> -- Binds_* -> count of fields (set/checked by check_binds_no_duplicate_fields)
|
--- @field seen_field table<string, integer> -- Binds_* -> count of fields (set/checked by check_binds_no_duplicate_fields)
|
||||||
--- @field _scan SourceScan -- full scan payload (typed-view sub-calls live here)
|
--- @field _scan SourceScan -- Full scan payload (typed-view sub-calls live here)
|
||||||
|
|
||||||
--- @class AnnotatedResult
|
--- @class AnnotatedResult
|
||||||
--- @field atoms AtomEntry[]
|
--- @field atoms AtomEntry[]
|
||||||
@@ -95,11 +95,10 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Per-check functions (the CHECK_RULES table's payload)
|
-- Per-check functions (the CHECK_RULES table's payload)
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
--
|
|
||||||
--- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
|
--- The dispatcher in `validate()` routes each result by convention: existence checks write errors[] and shape checks write warnings[].
|
||||||
--- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
|
--- `macro_word_drift` writes errors[] for missing or mismatched metadata and info[] for a match.
|
||||||
|
|
||||||
--- Check: every annotated atom must have a matching MipsAtom_(name) declaration.
|
--- Check: Every annotated atom must have a matching MipsAtom_(name) declaration.
|
||||||
--- @param a AtomAnnotation
|
--- @param a AtomAnnotation
|
||||||
--- @param pipe_ctx PipeCtx
|
--- @param pipe_ctx PipeCtx
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
@@ -112,8 +111,8 @@ local function check_atom_decl_exists(a, pipe_ctx, findings)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Check: every atom may have AT MOST ONE annotation.
|
--- Check: Every atom may have AT MOST ONE annotation.
|
||||||
--- Post-loop: needs full-corpus `annot_counts` from pipe_ctx.
|
--- Post-loop: Needs full-corpus `annot_counts` from pipe_ctx.
|
||||||
--- @param pipe_ctx PipeCtx
|
--- @param pipe_ctx PipeCtx
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_unique_annotation(pipe_ctx, findings)
|
local function check_unique_annotation(pipe_ctx, findings)
|
||||||
@@ -146,7 +145,7 @@ end
|
|||||||
--- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
|
--- Check: TAPE_WORDS(mac_X, N) ↔ WORD_COUNT(mac_X, N) drift.
|
||||||
--- Three outcomes: missing (error), mismatch (error), match (info).
|
--- Three outcomes: missing (error), mismatch (error), match (info).
|
||||||
--- @param m MacroEntry
|
--- @param m MacroEntry
|
||||||
--- @param wc table<string, integer> -- the shared word-count table (from ctx.shared.word_counts)
|
--- @param wc table<string, integer> -- Shared word-count table (from ctx.shared.word_counts)
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_macro_word_drift(m, wc, findings)
|
local function check_macro_word_drift(m, wc, findings)
|
||||||
local declared = wc[m.name]
|
local declared = wc[m.name]
|
||||||
@@ -304,7 +303,7 @@ local function check_binds_no_duplicate_fields(_src, pipe_ctx, findings)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Check: debug-skip markers must satisfy shape + placement constraints.
|
-- Check: Debug-skip markers must satisfy shape + placement constraints.
|
||||||
--- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
|
--- Walks the priority list once; each marker produces at most one error, so one source defect yields one finding.
|
||||||
--- Priority order (first defect wins):
|
--- Priority order (first defect wins):
|
||||||
--- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
|
--- 1. marker_kind ~= "atom_dbg_skip" -> legacy/renamed spelling (use `atom_dbg_skip`)
|
||||||
@@ -315,12 +314,11 @@ end
|
|||||||
--- 6. unsupported target_kind -> marker precedes an unrelated declaration
|
--- 6. unsupported target_kind -> marker precedes an unrelated declaration
|
||||||
--- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
|
--- Valid markers stamp `debug_skip` on whole-atom, bare-component, and proc-component declaration records in scan_source.lua.
|
||||||
--- @param marker DebugSkipMarker
|
--- @param marker DebugSkipMarker
|
||||||
--- @param _pipe_ctx PipeCtx -- unused today; kept for plex-shape consistency with per_annot
|
--- @param _pipe_ctx PipeCtx -- Unused; kept for consistency with per_annot // TODO(Ed): Remove?
|
||||||
--- @param findings Findings
|
--- @param findings Findings
|
||||||
local function check_skip_marker(marker, _pipe_ctx, findings)
|
local function check_skip_marker(marker, _pipe_ctx, findings)
|
||||||
local kind = marker.marker_kind
|
local kind = marker.marker_kind
|
||||||
local line = marker.marker_line
|
local line = marker.marker_line
|
||||||
|
|
||||||
-- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
|
-- Left `scan.debug_skip_markers` with production records for `atom_dbg_skip` only; other identifiers take the walker's unrelated branch.
|
||||||
|
|
||||||
if marker.has_parens then
|
if marker.has_parens then
|
||||||
@@ -371,8 +369,6 @@ local function check_skip_marker(marker, _pipe_ctx, findings)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Warn when a source references an unregistered alias.
|
--- Warn when a source references an unregistered alias.
|
||||||
---
|
|
||||||
--- R_TapePtr, R_AtomJmp, R_PrimCursor, R_FaceCursor, R_VertBase, and R_OtBase opt in through `#define atom_reg` in lottes_tape.h.
|
|
||||||
--- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
|
--- When a source uses an unregistered R_X, this check emits one pass-level info entry for that source and directs C-ABI register names to explicit alias registration.
|
||||||
--- @param _src SourceFile
|
--- @param _src SourceFile
|
||||||
--- @param pipe_ctx PipeCtx
|
--- @param pipe_ctx PipeCtx
|
||||||
@@ -426,8 +422,7 @@ local CHECK_RULES = {
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Validation
|
-- Validation
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
--
|
-- Pure check: Read from src.scan, run validations, emit findings. The scan was done once upstream.
|
||||||
-- Pure check: read from src.scan, run validations, emit findings. The scan was done once upstream.
|
|
||||||
|
|
||||||
--- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
|
--- Builds one pass-wide pipe_ctx from the merged `corpus.*` registries and source-ordered `corpus.atom_infos`; per-source declarations and bodies remain in `src.scan`.
|
||||||
--- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
|
--- The module ownership contract above requires callers to construct `ctx.shared.corpus` through `build_ctx`; the error message below enforces that gate.
|
||||||
@@ -473,7 +468,7 @@ end
|
|||||||
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
--- Validate one source against its pre-scanned SourceScan payload + the corpus-wide pipe_ctx.
|
||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @param src SourceFile
|
--- @param src SourceFile
|
||||||
--- @param corpus_pipe_ctx PipeCtx|nil -- built once per pass from corpus registries; nil builds the same projection here.
|
--- @param corpus_pipe_ctx PipeCtx|nil -- Built once per pass from corpus registries; nil builds the same projection here.
|
||||||
--- @return AnnotatedResult
|
--- @return AnnotatedResult
|
||||||
local function validate(ctx, src, corpus_pipe_ctx)
|
local function validate(ctx, src, corpus_pipe_ctx)
|
||||||
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
|
corpus_pipe_ctx = corpus_pipe_ctx or build_corpus_pipe_ctx(ctx)
|
||||||
@@ -503,14 +498,8 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
|
-- Build a per-source pipe_ctx: shared lookups come from `corpus_pipe_ctx`, while declarations, bodies, types, views, defaults, and occurrences come from `src.scan`.
|
||||||
local seen_defaults = {}
|
local seen_defaults = {}; for reg, _ in pairs (scan.types or {}) do seen_defaults[reg] = (seen_defaults[reg] or 0) + 1 end
|
||||||
for reg, _ in pairs(scan.types or {}) do
|
local atom_infos_list = {}; for _, ai in ipairs(scan.atom_infos or {}) do atom_infos_list[#atom_infos_list + 1] = ai end
|
||||||
seen_defaults[reg] = (seen_defaults[reg] or 0) + 1
|
|
||||||
end
|
|
||||||
local atom_infos_list = {}
|
|
||||||
for _, ai in ipairs(scan.atom_infos or {}) do
|
|
||||||
atom_infos_list[#atom_infos_list + 1] = ai
|
|
||||||
end
|
|
||||||
|
|
||||||
local pipe_ctx = {
|
local pipe_ctx = {
|
||||||
atom_index = {},
|
atom_index = {},
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
|
--- passes/atoms_source_map.lua — Per-.word source-line map emitter for tape atoms.
|
||||||
---
|
---
|
||||||
--- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
|
--- Writer: this pass, given `atom.paths` (the per-atom mutable surface owned by `emission_model`). Readers:
|
||||||
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and the gdb-runtime
|
--- `passes/dwarf_injection.lua` (synthesizes DW_TAG_inlined_subroutine + per-word line program rows) and
|
||||||
--- wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
|
--- the gdb-runtime wrapper at `scripts/gdb/gdb_tape_atoms.gdb` (loads the source map via `source <path>`).
|
||||||
---
|
---
|
||||||
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
|
--- Inputs from `atom.paths`: the ordered `items` stream, dense `word_events`, `invocations` views. Outputs:
|
||||||
--- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
--- one `WORD N LINE L TEXT T` line per emitted `.word`, plus the per-word provenance form that DWARF synthesis consumes.
|
||||||
@@ -28,7 +28,6 @@
|
|||||||
--- ...
|
--- ...
|
||||||
--- ENDATOM
|
--- ENDATOM
|
||||||
--- ```
|
--- ```
|
||||||
---
|
|
||||||
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
|
--- Marker records are zero-width in `atom.paths.items`, so they emit no WORD rows in the dense word view.
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -39,8 +38,8 @@
|
|||||||
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
|
-- (works both standalone + when require'd). `duffle_paths.lua` sets package.path then returns `require("duffle")`
|
||||||
-- at the bottom, so the dofile value IS the duffle module.
|
-- at the bottom, so the dofile value IS the duffle module.
|
||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
local elf_dwarf = require("elf_dwarf")
|
local elf_dwarf = require("elf_dwarf")
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Constants
|
-- Constants
|
||||||
@@ -69,8 +68,8 @@ local FORMAT_VERSION = 1
|
|||||||
--- @param atom table
|
--- @param atom table
|
||||||
--- @return table[], integer
|
--- @return table[], integer
|
||||||
local function canonical_word_entries(atom)
|
local function canonical_word_entries(atom)
|
||||||
local paths = atom.paths or {}
|
local paths = atom.paths or {}
|
||||||
local events = paths.word_events or {}
|
local events = paths.word_events or {}
|
||||||
local word_items = {}
|
local word_items = {}
|
||||||
for _, item in ipairs(paths.items or {}) do
|
for _, item in ipairs(paths.items or {}) do
|
||||||
if item.kind == "word" then word_items[#word_items + 1] = item end
|
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||||
@@ -95,41 +94,38 @@ end
|
|||||||
--- Render one atom's provenance stanza. Format 1 line shapes:
|
--- Render one atom's provenance stanza. Format 1 line shapes:
|
||||||
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
|
--- `WORD N CALL <src-path>:<src-line> MACRO <name> "<def-path>:<def-line>" BODY <line>` (component invocation)
|
||||||
--- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
|
--- `WORD N CALL <src-path>:<src-line> RAW` (raw `.word` outside any mac_* component)
|
||||||
--- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was
|
--- Component identity comes from the outermost invocation record; the count-table lookup confirms the component was declared in `corpus.word_counts`
|
||||||
--- declared in `corpus.word_counts` (populated by word_count_eval + components passes).
|
--- (populated by word_count_eval + components passes).
|
||||||
--- @param src table
|
--- @param src table
|
||||||
--- @param atom table
|
--- @param atom table
|
||||||
--- @param wc table -- identity alias of corpus.word_counts
|
--- @param wc table -- identity alias of corpus.word_counts
|
||||||
--- @return string[], integer
|
--- @return string[], integer
|
||||||
local function emit_provenance_stanza(src, atom, wc)
|
local function emit_provenance_stanza(src, atom, wc)
|
||||||
local lines = {}
|
local lines = {}
|
||||||
local rel_path = src.path:gsub("\\\\", "/")
|
local rel_path = src.path:gsub("\\\\", "/")
|
||||||
local entries, total = canonical_word_entries(atom)
|
local entries, total = canonical_word_entries(atom)
|
||||||
|
|
||||||
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||||
|
|
||||||
for _, entry in ipairs(entries) do
|
for _, entry in ipairs(entries) do
|
||||||
local inv = entry.invocation
|
local inv = entry.invocation
|
||||||
local macro_count = inv and wc["mac_" .. inv.component_name]
|
local macro_count = inv and wc["mac_" .. inv.component_name]
|
||||||
if inv and macro_count ~= nil then
|
if inv and macro_count ~= nil then
|
||||||
lines[#lines + 1] = string.format(
|
lines[#lines + 1] = string.format('WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d'
|
||||||
'WORD %d CALL %s:%d MACRO %s "%s:%d" BODY %d',
|
, entry.pos, rel_path, entry.line, inv.component_name
|
||||||
entry.pos, rel_path, entry.line, inv.component_name,
|
, inv.def_path or "", inv.def_line or 0, entry.body_line)
|
||||||
inv.def_path or "", inv.def_line or 0, entry.body_line)
|
|
||||||
else
|
else
|
||||||
lines[#lines + 1] = string.format(
|
lines[#lines + 1] = string.format("WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
||||||
"WORD %d CALL %s:%d RAW", entry.pos, rel_path, entry.line)
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
lines[1] = lines[1]:gsub(" 0$", " " .. tostring(total))
|
||||||
lines[#lines + 1] = "ENDATOM"
|
lines[#lines + 1] = "ENDATOM"
|
||||||
return lines, total
|
return lines, total
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Render the full provenance file content for one source.
|
--- Render the full provenance file content for one source.
|
||||||
--- @param src table
|
--- @param src table
|
||||||
--- @param wc table
|
--- @param wc table
|
||||||
--- @return string
|
--- @return string
|
||||||
local function render_provenance(src, wc)
|
local function render_provenance(src, wc)
|
||||||
local lines = {}
|
local lines = {}
|
||||||
@@ -162,8 +158,8 @@ end
|
|||||||
--- @param wc table
|
--- @param wc table
|
||||||
--- @return string[], integer
|
--- @return string[], integer
|
||||||
local function emit_atom_stanza(src, atom)
|
local function emit_atom_stanza(src, atom)
|
||||||
local lines = {}
|
local lines = {}
|
||||||
local rel_path = src.path:gsub("\\\\", "/")
|
local rel_path = src.path:gsub("\\\\", "/")
|
||||||
local entries, total = canonical_word_entries(atom)
|
local entries, total = canonical_word_entries(atom)
|
||||||
|
|
||||||
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
lines[#lines + 1] = string.format('ATOM %s "%s" 0', atom.raw_name or atom.name, rel_path)
|
||||||
@@ -179,10 +175,10 @@ end
|
|||||||
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
|
--- Render the full source map file content for one source (one .atoms.sourcemap.txt per source). Mirrors offsets.lua's
|
||||||
--- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
|
--- `project_atoms` shape: scan.atoms + scan.raw_atoms, no kind filter.
|
||||||
--- @param src table
|
--- @param src table
|
||||||
--- @param wc table
|
--- @param wc table
|
||||||
--- @return string
|
--- @return string
|
||||||
local function render_source_map(src)
|
local function render_source_map(src)
|
||||||
local lines = {}
|
local lines = {}
|
||||||
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
|
lines[#lines + 1] = "# FORMAT_VERSION " .. FORMAT_VERSION
|
||||||
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
lines[#lines + 1] = "# auto-generated by ps1_meta.lua (passes/atoms_source_map.lua) — DO NOT EDIT"
|
||||||
|
|
||||||
@@ -216,16 +212,16 @@ end
|
|||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries}
|
--- @return table[] -- list of {idx, name, src_path, file_base, addr, size_bytes, words, entries}
|
||||||
local function build_atom_table(ctx)
|
local function build_atom_table(ctx)
|
||||||
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
|
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
|
||||||
local corpus = ctx.shared and ctx.shared.corpus
|
local corpus = ctx.shared and ctx.shared.corpus
|
||||||
local matched = {}
|
local matched = {}
|
||||||
|
|
||||||
for _, src in ipairs(corpus.source_order or {}) do
|
for _, src in ipairs(corpus.source_order or {}) do
|
||||||
local file_base = src.path:match("([^/\\\\]+)$") or src.path
|
local file_base = src.path:match("([^/\\\\]+)$") or src.path
|
||||||
local function append(atom)
|
local function append(atom)
|
||||||
if not atom.paths then return end
|
if not atom.paths then return end
|
||||||
local name = atom.raw_name or atom.name
|
local name = atom.raw_name or atom.name
|
||||||
local info = addrs[name]
|
local info = addrs[name]
|
||||||
if not info then return end
|
if not info then return end
|
||||||
local entries, total = canonical_word_entries(atom)
|
local entries, total = canonical_word_entries(atom)
|
||||||
matched[#matched + 1] = {
|
matched[#matched + 1] = {
|
||||||
@@ -248,9 +244,10 @@ local function build_atom_table(ctx)
|
|||||||
return matched
|
return matched
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`, the convenience
|
--- Append the 9 gdb command definitions to `lines`. Pure gdb scripting — addresses come from `nm`,
|
||||||
--- vars set in `emit_gdb_runtime` provide printf args, and each command is a static sequence of `printf` / `tbreak` /
|
--- the convenience vars set in `emit_gdb_runtime` provide printf args, and
|
||||||
--- `if ... end` blocks. The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
|
--- each command is a static sequence of `printf` / `tbreak` / `if ... end` blocks.
|
||||||
|
--- The Lua pass emits N atoms' worth of lines; runtime iteration is gdb's job.
|
||||||
---
|
---
|
||||||
--- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
|
--- Why hardcoded per-atom: gdb's `$` substitution doesn't concat inside var names — `$__atom_name_$__i` in a `while`
|
||||||
--- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
|
--- loop resolves to one literal identifier, not `name_i`. Compile-time emission is the only path.
|
||||||
@@ -395,8 +392,7 @@ local function append_gdb_commands(lines, matched)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
|
--- Emit the gdb-runtime file (post-link). Pure gdb scripting — addresses come from `mipsel-none-elf-nm -S`, get embedded
|
||||||
--- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb
|
--- in `<ctx.out_root>/gdb_tape_atoms_runtime.gdb`, and load via `set $var = ...` + `define ... end` blocks at gdb source-time.
|
||||||
--- source-time.
|
|
||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
local function emit_gdb_runtime(ctx)
|
local function emit_gdb_runtime(ctx)
|
||||||
if not (ctx.flags and ctx.flags.gdb_runtime) then return end
|
if not (ctx.flags and ctx.flags.gdb_runtime) then return end
|
||||||
|
|||||||
@@ -0,0 +1,355 @@
|
|||||||
|
--- passes/auto_reg.lua — Per-phase automatic GPR allocator + gen/auto_reg.h emitter.
|
||||||
|
---
|
||||||
|
--- Reads the per-source + corpus-level `atom_auto_regs` + `phase_auto_regs` registries populated by `passes/scan_source.lua`.
|
||||||
|
--- Runs a deterministic first-fit allocator in the `R_T0..R_T7 + R_V0..R_V1` pool (10 physical GPRs).
|
||||||
|
--- Emits one `#define R_<Sym>_Code R_Tn_Code` per marker into per-directory `gen/auto_reg.h`.
|
||||||
|
---
|
||||||
|
--- User-pinned GPRs : The corpus's `register_alias_registry` is consulted to exclude GPRs the user has pinned via
|
||||||
|
--- `atom_reg` + `_Code` defs (e.g. carriers like `R_ResolveScratch = R_T4 atom_reg`).
|
||||||
|
--- These GPRs are unavailable to EVERY atom's source pool.
|
||||||
|
--- Carriers are preserved across atoms by context discipline and must never be reallocated.
|
||||||
|
--- Per-atom body parsing also catches alias references (R_<Alias>) and hardcoded R_Tn references,
|
||||||
|
--- so the user can write either `R_T4` or `R_ResolveScratch` in an atom body and the pass will
|
||||||
|
--- exclude R_T4 from that atom's pool.
|
||||||
|
---
|
||||||
|
--- Conflict detection: If the user hardcodes `R_Tn` in an atom body that shares a phase with an auto-reg that picked `R_Tn`,
|
||||||
|
--- emit `phase_register_clash` as an info finding (no build stop).
|
||||||
|
--- Should be unreachable after the user-pinning + body-parsing fix above; kept as a defensive safety net.
|
||||||
|
---
|
||||||
|
--- Pool exhaustion: If a phase declares more `R_<Sym>` mappings than the 10-register pool can hold,
|
||||||
|
--- emit `phase_register_pool_exhausted` as a build-stopping error.
|
||||||
|
|
||||||
|
--- @class AutoRegResult
|
||||||
|
--- @field outputs table[] -- {kind=, path=} entries
|
||||||
|
--- @field errors table[] -- {line=, msg=} entries (build-stops)
|
||||||
|
--- @field warnings table[] -- {line=, msg=} entries (build-continues)
|
||||||
|
|
||||||
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
|
||||||
|
--- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
--- THE GPR ALLOCATION POOL — what is allocatable, and (more importantly) WHY
|
||||||
|
--- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
---
|
||||||
|
--- The auto-reg pass picks physical GPRs for `atom_auto_reg(...)` / `phase_auto_reg(...)` markers.
|
||||||
|
--- It allocates from a FIXED 10-register pool.
|
||||||
|
--- This comment block makes the inclusion AND exclusion criteria obvious so a reader doesn't have
|
||||||
|
--- to grep lottes_tape.h + mips.h to understand the design.
|
||||||
|
---
|
||||||
|
--- ── WHAT'S IN THE POOL (10 GPRs, all caller-trash per the O32 ABI) ────────
|
||||||
|
--- R_T0..R_T7 (GPR codes 8..15), R_V0..R_V1 (GPR codes 2..3)
|
||||||
|
--- The workhorse of every atom body. The uesr should be aware of atom allocation across atoms they chain.
|
||||||
|
--- If they have a collision it means either they didn't saturate the register file optimally for a phase,
|
||||||
|
--- or the may have made the workload to large for the run.
|
||||||
|
---
|
||||||
|
--- ── WHAT'S NOT IN THE POOL — and WHY (the "obvious exclusions") ────────────
|
||||||
|
--- R_T9 (GPR code 25) — R_TapePtr, the tape instruction stream pointer.
|
||||||
|
--- Owned by the tape runtime (in tape_run / tape_run_a02_s07).
|
||||||
|
--- `rgcc(R_TapePtr)` register-variable ties the C compiler's view to $t9 across the whole tape_run.
|
||||||
|
--- The auto-reg pass MUST NOT clobber this; doing so would desync the C-side tape pointer from the
|
||||||
|
--- hardware pointer and crash on the next tape_run.
|
||||||
|
---
|
||||||
|
--- R_T8 (GPR code 24) — R_AtomJmp, the atom-jump register used by the 4-word yield handshake.
|
||||||
|
--- Every `mac_yield()` / `mac_yield_tail` does `load_word R_AtomJmp, R_TapePtr, 0` then
|
||||||
|
--- `jump_reg R_AtomJmp`. The auto-reg pass MUST NOT clobber this either, or the atom dispatcher breaks.
|
||||||
|
--- Owned by the tape runtime, same family as R_TapePtr.
|
||||||
|
---
|
||||||
|
--- R_AT (GPR code 1) — Assembler temporary. Reserved by the MIPS O32 ABI for pseudoinstruction expansion
|
||||||
|
--- (lottes_tape.h:86, mips.h:93). The ISA's psuedo instructions use it as a scratch temporary.
|
||||||
|
---
|
||||||
|
--- R_A0..A3 (codes 4..7) — Function arguments. Used in tape_run_a02_s07, see below.
|
||||||
|
--- R_S0..S7 (codes 16..23) — Callee-saved. Preserved across C-ABI calls by convention.
|
||||||
|
--- The `tape_run_a02_s07` variant clobbers them deliberately, but the default `tape_run` does NOT.
|
||||||
|
--- Kept out of POOL to preserve the conservative default.
|
||||||
|
--- Add them in a separate "big clobber" pool if/when needed.
|
||||||
|
---
|
||||||
|
--- R_K0/K1 (codes 26..27) — Kernel / interrupt handler reserves. Never touched by user code; OS-internal.
|
||||||
|
--- R_GP/SP/FP/RA (codes 28..31) — Stack frame + return-address. Owned by the C compiler; never allocatable.
|
||||||
|
--- R_0 (code 0) — Hardwired zero. Cannot be written.
|
||||||
|
---
|
||||||
|
local POOL = {
|
||||||
|
"R_T0", "R_T1", "R_T2", "R_T3",
|
||||||
|
"R_T4", "R_T5", "R_T6", "R_T7",
|
||||||
|
"R_V0", "R_V1",
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Map from integer MIPS GPR code (the `code` field on AliasEntry) to the physical GPR ident in POOL.
|
||||||
|
-- The standard MIPS O32 ABI register numbering matches mips.h's R_*_Code #defines (mips.h).
|
||||||
|
-- Only the POOL entries matter for auto_reg — non-pool aliases
|
||||||
|
-- (R_AT=1, R_A0..A3=4..7, R_T8=24, R_T9=25, R_K0/K1=26..27, R_GP/SP/FP/RA=28..31)
|
||||||
|
-- are deliberately omitted — see the comment block above for the WHY of each exclusion.
|
||||||
|
local INT_CODE_TO_POOL_GPR = {
|
||||||
|
[2] = "R_V0", [3] = "R_V1",
|
||||||
|
[8] = "R_T0", [9] = "R_T1", [10] = "R_T2", [11] = "R_T3",
|
||||||
|
[12] = "R_T4", [13] = "R_T5", [14] = "R_T6", [15] = "R_T7",
|
||||||
|
}
|
||||||
|
|
||||||
|
-- Stable sort for deterministic allocation order.
|
||||||
|
local function stable_sort_keys(tbl)
|
||||||
|
local keys = {}
|
||||||
|
for k in pairs(tbl) do keys[#keys + 1] = k end
|
||||||
|
table.sort(keys)
|
||||||
|
return keys
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Allocate one phase's auto-reg mappings.
|
||||||
|
-- Returns (allocated_map, errors). On pool exhaustion, errors is populated and the function halts.
|
||||||
|
local function allocate_phase(phase_label, decls)
|
||||||
|
-- Deep-copy POOL into a fresh sequence table. The original `table.unpack and table.unpack(POOL) or { unpack(POOL) }`
|
||||||
|
-- idiom wraps the unpacked values in a single inner table under LuaJIT 5.1 (`table.unpack` is nil; the `or` returns one value),
|
||||||
|
-- which corrupts the pool into `{ {R_T0, R_T1, ...} }` — making `table.remove(pool, 1)` return the inner table on iteration.
|
||||||
|
local pool = {}
|
||||||
|
for i = 1, #POOL do pool[i] = POOL[i] end
|
||||||
|
local result = {}
|
||||||
|
local errors = {}
|
||||||
|
for _, sym in ipairs(stable_sort_keys(decls)) do
|
||||||
|
local next_gpr = table.remove(pool, 1)
|
||||||
|
if not next_gpr then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = string.format("phase_register_pool_exhausted: "
|
||||||
|
.. "phase '%s' requested symbol '%s' but the pool has no remaining registers "
|
||||||
|
.. "(max 10 per phase: R_T0..R_T7 + R_V0..R_V1). Split the phase or use hardcoded GPRs."
|
||||||
|
, phase_label, sym),
|
||||||
|
}
|
||||||
|
return result, errors
|
||||||
|
end
|
||||||
|
result[sym] = next_gpr
|
||||||
|
end
|
||||||
|
return result, errors
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Build two projections from corpus.register_alias_registry:
|
||||||
|
-- user_pinned -- { [physical_gpr_ident] = true } -- GPRs unavailable to auto_reg globally (wave-context carriers, file-scope pinned aliases)
|
||||||
|
-- alias_to_gpr -- { [alias_ident] = physical_gpr_ident } -- for body parsing
|
||||||
|
-- Both projections are derived from the same set of entries: every AliasEntry in register_alias_registry has `has_atom_reg = true`
|
||||||
|
-- (only those entries are added to the registry; see passes/scan_source.lua parse_enum_entry).
|
||||||
|
-- Each entry's `code` is the integer MIPS GPR number (0..31); INT_CODE_TO_POOL_GPR translates it back to the physical GPR ident.
|
||||||
|
-- Aliases whose `code` points to a non-POOL GPR (e.g. R_S0, R_T8, R_K1) are ignored —
|
||||||
|
-- they don't affect the auto_reg pool, and they're already excluded from POOL above.
|
||||||
|
local function build_user_pins(corpus)
|
||||||
|
local user_pinned = {}
|
||||||
|
local alias_to_gpr = {}
|
||||||
|
if not corpus.register_alias_registry then return user_pinned, alias_to_gpr end
|
||||||
|
for alias_name, alias_entry in pairs(corpus.register_alias_registry) do
|
||||||
|
if alias_entry.has_atom_reg and alias_entry.code then
|
||||||
|
local gpr = INT_CODE_TO_POOL_GPR[alias_entry.code]
|
||||||
|
if gpr then
|
||||||
|
user_pinned[gpr] = true
|
||||||
|
alias_to_gpr[alias_name] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return user_pinned, alias_to_gpr
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Find every physical GPR referenced in the atom body, via EITHER:
|
||||||
|
-- (a) A hardcoded physical GPR ident (R_T\d+|R_V\d+|R_A\d+|R_S\d+) — the existing regex;
|
||||||
|
-- (b) An alias ident (R_<Alias>) resolved via alias_to_gpr back to its physical GPR ident.
|
||||||
|
-- Returns { [physical_gpr_ident] = count }. Clash-detection and source-pool-exclusion logic
|
||||||
|
-- only needs the presence of each GPR (boolean test), but keeping count preserves the
|
||||||
|
-- original find_hardcoded_rn shape so callers can switch without churn.
|
||||||
|
-- The alias pattern is sorted lexicographically to keep the regex deterministic.
|
||||||
|
local function find_used_gprs(body_text, alias_to_gpr)
|
||||||
|
local found = {}
|
||||||
|
-- (a) Hardcoded physical GPRs (R_T0..R_T7, R_V0..R_V1, R_A0..R_A3, R_S0..R_S7).
|
||||||
|
for gpr in body_text:gmatch("(R_T%d+|R_V%d+|R_A%d+|R_S%d+)") do
|
||||||
|
found[gpr] = (found[gpr] or 0) + 1
|
||||||
|
end
|
||||||
|
-- (b) Alias references (R_<Alias>) resolved to physical GPRs via the registry.
|
||||||
|
-- Sorted by name so the regex is byte-stable across runs.
|
||||||
|
if alias_to_gpr and next(alias_to_gpr) then
|
||||||
|
local aliases = {}
|
||||||
|
for alias_name in pairs(alias_to_gpr) do
|
||||||
|
aliases[#aliases + 1] = alias_name
|
||||||
|
end
|
||||||
|
table.sort(aliases)
|
||||||
|
local pattern = "(" .. table.concat(aliases, "|") .. ")"
|
||||||
|
for alias_name in body_text:gmatch(pattern) do
|
||||||
|
local gpr = alias_to_gpr[alias_name]
|
||||||
|
if gpr and not found[gpr] then
|
||||||
|
found[gpr] = 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return found
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Emit one gen/auto_reg.h header per directory.
|
||||||
|
local function emit_auto_reg_h(out_dir, dir, sources, mappings)
|
||||||
|
if not mappings or next(mappings) == nil then return end
|
||||||
|
local out_path = out_dir .. "/" .. "auto_reg.h"
|
||||||
|
duffle.ensure_dir(out_dir)
|
||||||
|
local lines = {
|
||||||
|
"#ifdef INTELLISENSE_DIRECTIVES",
|
||||||
|
"#pragma once",
|
||||||
|
"#endif",
|
||||||
|
"// Auto-generated by ps1_meta.lua (passes/auto_reg.lua) — DO NOT EDIT",
|
||||||
|
"// Directory: " .. dir:gsub("/", "\\"),
|
||||||
|
}
|
||||||
|
for _, src in ipairs(sources) do
|
||||||
|
lines[#lines + 1] = "// source: " .. src.path
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = "// Per-phase register allocations resolved by the lua pass."
|
||||||
|
lines[#lines + 1] = "// R_<Sym>_Code = <chosen GPR's _Code constant> for every marker in this directory."
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
for _, sym in ipairs(stable_sort_keys(mappings)) do
|
||||||
|
local gpr = mappings[sym]
|
||||||
|
local gpr_code = gpr .. "_Code"
|
||||||
|
lines[#lines + 1] = "#define " .. sym .. "_Code " .. gpr_code
|
||||||
|
end
|
||||||
|
lines[#lines + 1] = ""
|
||||||
|
duffle.write_file_lf(out_path, table.concat(lines, "\n") .. "\n")
|
||||||
|
print(" -> " .. out_path)
|
||||||
|
return out_path
|
||||||
|
end
|
||||||
|
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- Pass entry
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
local M = {}
|
||||||
|
|
||||||
|
--- @param ctx PassCtx
|
||||||
|
--- @return AutoRegResult
|
||||||
|
function M.run(ctx)
|
||||||
|
local outputs = {}
|
||||||
|
local errors = {}
|
||||||
|
local warnings = {}
|
||||||
|
|
||||||
|
local corpus = ctx.shared and ctx.shared.corpus
|
||||||
|
if type(corpus) ~= "table" then
|
||||||
|
error("auto_reg.run requires ctx.shared.corpus", 0)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 0. Build the user-pinned GPR exclusion set + alias-to-GPR resolution map.
|
||||||
|
-- Wave-context carriers (e.g. `R_ResolveScratch = R_T4 atom_reg` in hello_camera.atom.c)
|
||||||
|
-- MUST NOT be allocated to any auto-reg marker — they're preserved across atoms by the wave-context discipline.
|
||||||
|
-- The corpus's register_alias_registry is the source of truth for these opt-in pins.
|
||||||
|
-- Body references to those aliases (via alias_to_gpr) are also excluded on a per-atom basis in step 2 below.
|
||||||
|
local user_pinned, alias_to_gpr = build_user_pins(corpus)
|
||||||
|
|
||||||
|
-- 1. Allocate phase pools first (phase declarations take precedence over per-atom declarations).
|
||||||
|
local phase_allocations = {}
|
||||||
|
for phase_label, decls in pairs(corpus.phase_auto_regs or {}) do
|
||||||
|
local mapping, errs = allocate_phase(phase_label, decls)
|
||||||
|
for sym, gpr in pairs(mapping) do
|
||||||
|
phase_allocations[phase_label] = phase_allocations[phase_label] or {}
|
||||||
|
phase_allocations[phase_label][sym] = gpr
|
||||||
|
end
|
||||||
|
for _, e in ipairs(errs) do
|
||||||
|
errors[#errors + 1] = e
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 2. Allocate per-atom auto-regs. If the atom scope matches a phase, reuse the phase pool.
|
||||||
|
-- Otherwise, allocate a private pool for the atom.
|
||||||
|
-- The phase membership is in `corpus.atom_phases[phase_label].atoms` (an array of atom names declared via `atom_phase(<phase>)`
|
||||||
|
-- in the atom's `atom_info` line). Build a reverse map `atom_name -> phase_label` so the lookup is O(1) per atom scope.
|
||||||
|
local atom_name_to_phase = {}
|
||||||
|
for phase_label, entry in pairs(corpus.atom_phases or {}) do
|
||||||
|
for _, atom_name in ipairs(entry.atoms or {}) do
|
||||||
|
atom_name_to_phase[atom_name] = phase_label
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local atom_allocations = {}
|
||||||
|
for atom_scope, decls in pairs(corpus.atom_auto_regs or {}) do
|
||||||
|
local phase_label = atom_name_to_phase[atom_scope]
|
||||||
|
-- Build the atom's source pool: start with the full POOL, subtract:
|
||||||
|
-- (a) every GPR already committed (phase allocations + prior atom allocations)
|
||||||
|
-- (b) every USER-PINNED GPR (wave-context carriers + file-scope pinned aliases)
|
||||||
|
-- (c) every GPR referenced in the atom's body — either hardcoded R_X or alias R_Xxx
|
||||||
|
-- (the latter resolved via alias_to_gpr; this catches cases where the user wrote R_ResolveScratch instead of R_T4 directly)
|
||||||
|
-- Atoms whose scope matches a phase share the global pool with the phase allocations;
|
||||||
|
-- the original `source_pool = phase_allocations[phase_label]` form used the phase
|
||||||
|
-- allocation MAP as a pool, but that map has no array part, so `table.remove(source_pool, 1)`
|
||||||
|
-- returned nil and every atom-with-phase marker errored with `phase_register_pool_exhausted`.
|
||||||
|
local used = {}
|
||||||
|
for _, m in pairs(phase_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
|
||||||
|
for _, m in pairs(atom_allocations) do for _, gpr in pairs(m) do used[gpr] = true end end
|
||||||
|
-- (c) Body references — scan the atom body for hardcoded + alias-resolved GPRs.
|
||||||
|
-- Folded into `used` so the source_pool exclusion is a single check.
|
||||||
|
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
|
||||||
|
if atom and atom.body then
|
||||||
|
local body_used = find_used_gprs(atom.body, alias_to_gpr)
|
||||||
|
for gpr in pairs(body_used) do used[gpr] = true end
|
||||||
|
end
|
||||||
|
local source_pool = {}
|
||||||
|
for _, gpr in ipairs(POOL) do
|
||||||
|
-- Exclude (a) prior commitments, (b) USER-PINNED GPRs (wave-context carriers
|
||||||
|
-- declared via atom_reg + _Code defs, preserved across atoms globally).
|
||||||
|
if not used[gpr] and not user_pinned[gpr] then
|
||||||
|
source_pool[#source_pool + 1] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local result = {}
|
||||||
|
for _, sym in ipairs(stable_sort_keys(decls)) do
|
||||||
|
local next_gpr = table.remove(source_pool, 1)
|
||||||
|
if not next_gpr then
|
||||||
|
errors[#errors + 1] = {
|
||||||
|
line = 0,
|
||||||
|
msg = string.format("phase_register_pool_exhausted: atom '%s' requested symbol '%s' "
|
||||||
|
.. "but no free registers remain in its scope pool."
|
||||||
|
, atom_scope, sym),
|
||||||
|
}
|
||||||
|
else
|
||||||
|
result[sym] = next_gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
atom_allocations[atom_scope] = result
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 3. Conflict-with-hardcoded detection (defensive — should be unreachable now).
|
||||||
|
-- The source_pool exclusion in step 2 (b) + (c) already accounts for both user-pinned GPRs
|
||||||
|
-- and body-referenced GPRs (hardcoded R_Tn OR alias R_<Alias>).
|
||||||
|
-- An auto-reg allocation that matched an existing body reference would be impossible by construction.
|
||||||
|
-- This warning is kept as a defensive safety net for cases the body scanner might miss
|
||||||
|
-- (e.g. macros that expand to register references the scanner cannot resolve).
|
||||||
|
-- For each resolved (scope, sym) -> R_Tn mapping, scan the atom body source for used GPRs.
|
||||||
|
for atom_scope, decls in pairs(atom_allocations) do
|
||||||
|
local atom = corpus.atoms_by_name and corpus.atoms_by_name[atom_scope]
|
||||||
|
if atom and atom.body then
|
||||||
|
local used_in_body = find_used_gprs(atom.body, alias_to_gpr)
|
||||||
|
for sym, allocated_gpr in pairs(decls) do
|
||||||
|
if used_in_body[allocated_gpr] and used_in_body[allocated_gpr] > 0 then
|
||||||
|
warnings[#warnings + 1] = {
|
||||||
|
line = atom.line or 0,
|
||||||
|
msg = string.format("phase_register_clash: atom '%s' has hardcoded '%s' in its body AND an auto-reg marker '%s' "
|
||||||
|
.. "that was allocated to '%s' (same phase). Resolve by removing the hardcoded reference or renaming the auto-reg."
|
||||||
|
, atom_scope, allocated_gpr, sym, allocated_gpr),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- 4. Emit per-directory gen/auto_reg.h.
|
||||||
|
-- For each source directory that has atom_auto_regs or phase_auto_regs entries, emit one header.
|
||||||
|
local sources_by_dir = corpus.sources_by_dir or {}
|
||||||
|
for dir, sources in pairs(sources_by_dir) do
|
||||||
|
local per_dir_mappings = {}
|
||||||
|
for _, src in ipairs(sources) do
|
||||||
|
-- Collect every (sym -> gpr) entry that originated from a source in this directory.
|
||||||
|
-- `src.scan.atom_auto_regs` is keyed by ATOM SCOPE NAME; `pairs(t)` iterates KEYS so `scope_name` here is the scope ident (e.g. "cube_g4_face").
|
||||||
|
-- The previous `for _, scan_atom_auto` form silently assigned the VALUE (a `{sym = sym}` table) to the variable,
|
||||||
|
-- which made `atom_allocations[scan_atom_auto]` a table-indexed lookup that never resolved.
|
||||||
|
for scope_name in pairs(src.scan and src.scan.atom_auto_regs or {}) do
|
||||||
|
for sym, gpr in pairs(atom_allocations[scope_name] or {}) do
|
||||||
|
per_dir_mappings[sym] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for scope_name in pairs(src.scan and src.scan.phase_auto_regs or {}) do
|
||||||
|
for sym, gpr in pairs(phase_allocations[scope_name] or {}) do
|
||||||
|
per_dir_mappings[sym] = gpr
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local out_dir = dir .. "/gen"
|
||||||
|
local out_path = emit_auto_reg_h(out_dir, dir, sources, per_dir_mappings)
|
||||||
|
if out_path then outputs[#outputs + 1] = { auto_reg_h = out_path } end
|
||||||
|
end
|
||||||
|
return { outputs = outputs, errors = errors, warnings = warnings }
|
||||||
|
end
|
||||||
|
|
||||||
|
return M
|
||||||
+135
-118
@@ -3,13 +3,15 @@
|
|||||||
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
|
--- Ownership: `corpus.word_counts`, `corpus.components`, and `corpus.component_body_index`.
|
||||||
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
--- Scanner owns `declaration_comment` and `debug_skip` on each declaration record; this pass projects both forward.
|
||||||
---
|
---
|
||||||
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations,
|
--- Reads the pre-scanned SourceScan payload from `duffle.scan_source` for `MipsAtomComp_(ac_X)` and `MipsAtomComp_Proc_(ac_X, { body })` declarations (kind="comp_bare" / "comp_proc"),
|
||||||
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
--- then resolves the function-args string from the preceding `FI_ Slice_MipsCode ac_X(...)` declaration via a backward walk.
|
||||||
---
|
---
|
||||||
--- Emits one `<dir_basename>.macs.h` per source with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
--- `MipsAtom_Proc_(X, ab, { body })` declarations (kind="atom_proc") are ATOMS, not components, and are deliberately excluded —
|
||||||
|
--- atoms get emitted via `tb_emit(tb, code_<name>)` linker symbols, not inlined as `mac_*` macros.
|
||||||
---
|
---
|
||||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
--- Emits one `gen/macs.h` per *immediate source directory* with `#define mac_X(sig) \` macros plus `WORD_COUNT(mac_X, N)` entries for downstream offset computation.
|
||||||
--- Lua 5.3 compatible.
|
--- All sources inside the same directory contribute to the same file (per-directory aggregation).
|
||||||
|
--- The directory itself is the namespace, so the filename does not repeat the module name.
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Module-scope requires + package.path setup
|
-- Module-scope requires + package.path setup
|
||||||
@@ -41,43 +43,44 @@ local MAC_PREFIX_LEN = 4
|
|||||||
local BYTE_NEWLINE = 10
|
local BYTE_NEWLINE = 10
|
||||||
local BYTE_SLASH = 47
|
local BYTE_SLASH = 47
|
||||||
|
|
||||||
-- Source dir basename used as the output `.macs.h` filename.
|
-- Output gen subdirectory + filename (per-directory aggregation; the directory name is the namespace).
|
||||||
local GEN_SUBDIR = "gen"
|
local GEN_SUBDIR = "gen"
|
||||||
|
local MACS_FILENAME = "macs.h"
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Type declarations
|
-- Type declarations
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
--- @field scan table -- Pre-scanned SourceScan payload (from duffle.scan_source)
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field sources SourceFile[] -- all source files in the build
|
--- @field sources SourceFile[] -- All source files in the build
|
||||||
--- @field metadata_path string -- path to word_count.metadata.h
|
--- @field metadata_path string -- Path to word_count.metadata.h
|
||||||
--- @field shared table -- cross-pass shared state
|
--- @field shared table -- Cross-pass shared state
|
||||||
--- @field out_root string -- output root (e.g. "build/gen")
|
--- @field out_root string -- Output root (e.g. "build/gen")
|
||||||
--- @field project_root string -- project root (e.g. "code/")
|
--- @field project_root string -- Project root (e.g. "code/")
|
||||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
--- @field upstream table<string, table> -- Per-pass upstream outputs
|
||||||
--- @field flags table -- CLI flags
|
--- @field flags table -- CLI flags
|
||||||
--- @field verbose boolean -- log diagnostic info
|
--- @field verbose boolean -- Log diagnostic info
|
||||||
|
|
||||||
--- @class PassResult
|
--- @class PassResult
|
||||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||||
|
|
||||||
--- @class Component
|
--- @class Component
|
||||||
--- @field name string -- atom name (without `ac_` prefix)
|
--- @field name string -- Atom name (without `ac_` prefix)
|
||||||
--- @field body string -- brace-delimited body (without the braces)
|
--- @field body string -- Brace-delimited body (without the braces)
|
||||||
--- @field args string|nil -- function-args string (function form only)
|
--- @field args string|nil -- Function-args string (function form only)
|
||||||
--- @field line integer -- source line of the declaration
|
--- @field line integer -- Source line of the declaration
|
||||||
--- @field comment string|nil -- scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
--- @field comment string|nil -- Scanner-owned `declaration_comment`; the components pass reads it from the scanner record
|
||||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component — see `project_components`)
|
||||||
--- @field debug_skip boolean -- mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
--- @field debug_skip boolean -- Mirror of `a.debug_skip` (scanner-owned); true iff a bare `atom_dbg_skip` marker immediately preceded the declaration
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Local helpers (file I/O + path normalization)
|
-- Local helpers (file I/O + path normalization)
|
||||||
@@ -93,48 +96,21 @@ local M = {}
|
|||||||
-- so this file reads it forward rather than re-walking the source.
|
-- so this file reads it forward rather than re-walking the source.
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation of the given name.
|
--- Find the args of the function declaration that immediately precedes a `MipsAtomComp_Proc_` invocation.
|
||||||
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
--- Returns the args string (e.g., `"U4 off, U4 code, U1 r, U1 g, U1 b"`) or nil if no function declaration is found.
|
||||||
---
|
---
|
||||||
--- Convention: function form is
|
--- After the `sym` arg was dropped from MipsAtomComp_Proc_, the component name
|
||||||
--- `FI_ Slice_MipsCode ac_X(args) MipsAtomComp_Proc_(ac_X, { body })`
|
--- and the args both come from the preceding `FI_ Slice_MipsCode ac_X(args)`
|
||||||
--- We find the LAST occurrence of `"ac_X("` before `before_pos` and extract the args from inside the parens.
|
--- declaration. The shared `duffle.find_function_decl_for` helper does the
|
||||||
--- We then verify the preceding context ends with `Slice_MipsCode`
|
--- backward walk; this function returns just the args.
|
||||||
--- (the function-decl keyword with possible qualifiers between).
|
|
||||||
---
|
---
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param name string
|
--- @param name string (retained for signature stability; unused — the walk derives the name)
|
||||||
--- @param before_pos integer
|
--- @param before_pos integer
|
||||||
--- @return string|nil
|
--- @return string|nil
|
||||||
local function find_function_args_for(source, name, before_pos)
|
local function find_function_args_for(source, name, before_pos)
|
||||||
-- Find the LAST occurrence of `name + "("` in `source[1..before_pos]`.
|
local _, args_inner = duffle.find_function_decl_for(source, before_pos, #MIPS_ATOM)
|
||||||
local name_open = name .. "("
|
return args_inner
|
||||||
local last_idx = nil
|
|
||||||
local scan_pos = 1
|
|
||||||
while true do
|
|
||||||
-- Pass `before_pos + 1` so string.find only returns positions < before_pos + 1
|
|
||||||
-- (string.find's 4th arg `plain` is true; we use the 3rd arg `init` for the upper bound).
|
|
||||||
local found = source:find(name_open, scan_pos, true)
|
|
||||||
if not found or found >= before_pos then break end
|
|
||||||
last_idx = found
|
|
||||||
scan_pos = found + #name_open
|
|
||||||
end
|
|
||||||
if not last_idx then return nil end
|
|
||||||
|
|
||||||
-- Verify the preceding context ends with "MipsAtom" (with possible qualifiers between).
|
|
||||||
local before = source:sub(1, last_idx - 1)
|
|
||||||
local trimmed = duffle.trim(before)
|
|
||||||
if trimmed:sub(-#MIPS_ATOM) ~= MIPS_ATOM then
|
|
||||||
-- Preceding context is not a function declaration.
|
|
||||||
return nil
|
|
||||||
end
|
|
||||||
|
|
||||||
local open_paren = last_idx + #name -- position of "("
|
|
||||||
-- scan: MipsAtom ac_X(
|
|
||||||
local inner = duffle.read_parens(source, open_paren)
|
|
||||||
-- scan: MipsAtom ac_X(<args>)
|
|
||||||
if not inner then return nil end
|
|
||||||
return inner
|
|
||||||
end
|
end
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -200,7 +176,16 @@ end
|
|||||||
local function project_components(source, scan)
|
local function project_components(source, scan)
|
||||||
local out = {}
|
local out = {}
|
||||||
for _, a in ipairs(scan.atoms) do
|
for _, a in ipairs(scan.atoms) do
|
||||||
|
-- Only `MipsAtomComp_(ac_X)` (kind="comp_bare") and `MipsAtomComp_Proc_(ac_X, ...)` (kind="comp_proc")
|
||||||
|
-- are COMPONENTS — they get inlined via `mac_<name>` aliases inside atom bodies.
|
||||||
|
-- `MipsAtom_Proc_` (kind="atom_proc") is an ATOM (ends with `mac_yield()`); it gets emitted via
|
||||||
|
-- `tb_emit(tb, code_<name>)` (linker symbol), NOT inlined as a macro. Including `atom_proc` here
|
||||||
|
-- would incorrectly emit `mac_<name>` aliases for atoms, polluting `gen/macs.h`.
|
||||||
|
-- See `docs/duffle_dsl_primer.md` §"mac_* aliases" for the contract.
|
||||||
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
if a.kind == "comp_bare" or a.kind == "comp_proc" then
|
||||||
|
-- Function-args lookup is meaningful for `MipsAtomComp_Proc_` components
|
||||||
|
-- (the macro sits inside `FI_ Slice_MipsCode ac_X(...)`); the alias expansion
|
||||||
|
-- discards the `ab` (atom-builder) arg the same way both forms do.
|
||||||
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
local args = find_function_args_for(source, a.raw_name, a.ident_pos)
|
||||||
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
-- Comment ownership: scan_source.lua stamps `declaration_comment` on the record by walking backward past any associated bare marker.
|
||||||
-- The pass reads `declaration_comment` directly.
|
-- The pass reads `declaration_comment` directly.
|
||||||
@@ -231,7 +216,6 @@ end
|
|||||||
--
|
--
|
||||||
-- Skips `//` sequences that are inside string or character literals
|
-- Skips `//` sequences that are inside string or character literals
|
||||||
-- (a rough heuristic — sufficient for component bodies which don't have those constructs).
|
-- (a rough heuristic — sufficient for component bodies which don't have those constructs).
|
||||||
--
|
|
||||||
--- @param s string
|
--- @param s string
|
||||||
--- @return string
|
--- @return string
|
||||||
local function convert_line_comments_to_block(s)
|
local function convert_line_comments_to_block(s)
|
||||||
@@ -300,7 +284,9 @@ local function word_count_rec(name, comp_by_name, wc, cache)
|
|||||||
local trimmed = t.tok
|
local trimmed = t.tok
|
||||||
if trimmed ~= "" then
|
if trimmed ~= "" then
|
||||||
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
local lookup = strip_mac_prefix(duffle.read_ident(trimmed, 1))
|
||||||
if lookup and comp_by_name[lookup] then
|
if lookup == "atom_label" or lookup == "atom_offset" then
|
||||||
|
-- Pure metaprogram anchors; emit zero words.
|
||||||
|
elseif lookup and comp_by_name[lookup] then
|
||||||
-- It's a `mac_X(...)` call. Recurse.
|
-- It's a `mac_X(...)` call. Recurse.
|
||||||
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
n = n + word_count_rec(lookup, comp_by_name, wc, cache)
|
||||||
elseif lookup and wc and wc[lookup] then
|
elseif lookup and wc and wc[lookup] then
|
||||||
@@ -350,7 +336,6 @@ end
|
|||||||
|
|
||||||
--- (internal) Recursive cycle-cost derivation. Sum `latency[ident]` per emitted instruction in the component body,
|
--- (internal) Recursive cycle-cost derivation. Sum `latency[ident]` per emitted instruction in the component body,
|
||||||
--- recursing through nested `mac_*` calls (so `mac_format_g4_color`'s cost = 4 × `mac_pack_color_word`'s cost).
|
--- recursing through nested `mac_*` calls (so `mac_format_g4_color`'s cost = 4 × `mac_pack_color_word`'s cost).
|
||||||
---
|
|
||||||
--- Special rule: `mac_yield`'s cost = 0 (per `lottes_tape.h:125-130` "the runtime cost lands in the next atom's prologue").
|
--- Special rule: `mac_yield`'s cost = 0 (per `lottes_tape.h:125-130` "the runtime cost lands in the next atom's prologue").
|
||||||
--- @param name string -- component bare name (e.g. "yield", "pack_color_word")
|
--- @param name string -- component bare name (e.g. "yield", "pack_color_word")
|
||||||
--- @param comp_by_name table<string, Component>
|
--- @param comp_by_name table<string, Component>
|
||||||
@@ -392,9 +377,7 @@ local function cycle_cost_rec(name, comp_by_name, latency, cache)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
|
--- (internal) Recursive GP0 prim-buffer contribution. Count `store_word` / `store_half` / `store_byte`
|
||||||
--- calls in the component body that target `R_PrimCursor` (these are the
|
--- calls in the component body that target `R_PrimCursor` (these are the RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
|
||||||
--- RAM-side prim-buffer words the macro contributes), recursing through nested `mac_*` calls.
|
|
||||||
---
|
|
||||||
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
|
--- Only `R_PrimCursor`-targeting stores count. Stores targeting other registers (e.g. `R_OtBase`, heap pointers) are not prim-buffer contributions.
|
||||||
--- @param name string
|
--- @param name string
|
||||||
--- @param comp_by_name table<string, Component>
|
--- @param comp_by_name table<string, Component>
|
||||||
@@ -410,9 +393,9 @@ local function gp0_contrib_rec(name, comp_by_name, cache)
|
|||||||
local tokens = cc.body_tokens
|
local tokens = cc.body_tokens
|
||||||
for _, t in ipairs(tokens) do
|
for _, t in ipairs(tokens) do
|
||||||
local trimmed = t.tok
|
local trimmed = t.tok
|
||||||
if trimmed ~= "" then
|
if trimmed ~= "" then
|
||||||
local ident = duffle.read_ident(trimmed, 1)
|
local ident = duffle.read_ident(trimmed, 1)
|
||||||
if ident and ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
|
if ident and ident:sub(1, MAC_PREFIX_LEN) == MAC_PREFIX then
|
||||||
-- Nested `mac_X(...)` call: recurse.
|
-- Nested `mac_X(...)` call: recurse.
|
||||||
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
local nested = ident:sub(MAC_PREFIX_LEN + 1)
|
||||||
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
n = n + gp0_contrib_rec(nested, comp_by_name, cache)
|
||||||
@@ -460,8 +443,8 @@ end
|
|||||||
--- @param s string
|
--- @param s string
|
||||||
--- @return string[]
|
--- @return string[]
|
||||||
local function split_comment_lines(s)
|
local function split_comment_lines(s)
|
||||||
local out = {}
|
local out = {}
|
||||||
local pos = 1
|
local pos = 1
|
||||||
local s_len = #s
|
local s_len = #s
|
||||||
while pos <= s_len do
|
while pos <= s_len do
|
||||||
local nl = s:find("\n", pos, true)
|
local nl = s:find("\n", pos, true)
|
||||||
@@ -476,12 +459,25 @@ local function split_comment_lines(s)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
--- Determine the macro signature: function-args list (function form) or variadic-ignored (bare form).
|
||||||
|
--- For `MipsAtomComp_Proc_` components, the leading `ab` (atom-builder) arg is dropped:
|
||||||
|
--- the generated `mac_<name>` macros are inline-expansion aliases for baked atoms; their bodies don't reference `ab`
|
||||||
|
--- (the builder is only consumed by the procedural `atombuilder_unroll` line that `MipsAtomComp_Proc_` appends after the body).
|
||||||
|
--- Inline callers therefore don't need to thread a builder context.
|
||||||
--- @param args_str string|nil
|
--- @param args_str string|nil
|
||||||
--- @return string
|
--- @return string
|
||||||
local function signature_from_args(args_str)
|
local function signature_from_args(args_str)
|
||||||
local arg_names = extract_arg_names(args_str)
|
local arg_names = extract_arg_names(args_str)
|
||||||
if arg_names and #arg_names > 0 then
|
if arg_names and #arg_names > 0 then
|
||||||
return table.concat(arg_names, ", ")
|
-- Drop the leading `ab` (atom-builder) first arg if present.
|
||||||
|
-- Convention: `MipsAtomComp_Proc_` components always declare `ab` as the first function-arg
|
||||||
|
-- (type `MipsAtomBuilder_R`), mirroring the macro signature in `lottes_tape.h`.
|
||||||
|
if arg_names[1] == "ab" then
|
||||||
|
table.remove(arg_names, 1)
|
||||||
|
end
|
||||||
|
if #arg_names > 0 then
|
||||||
|
return table.concat(arg_names, ", ")
|
||||||
|
end
|
||||||
|
return "..." -- `ab` was the only arg; fall through to variadic
|
||||||
end
|
end
|
||||||
return "..."
|
return "..."
|
||||||
end
|
end
|
||||||
@@ -523,7 +519,7 @@ local function build_component_lines(c, counts)
|
|||||||
|
|
||||||
-- Marker comment: emitted once for every skipped component.
|
-- Marker comment: emitted once for every skipped component.
|
||||||
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
|
-- The marker is scanner-owned (declared by `atom_dbg_skip` immediately before the declaration in the source);
|
||||||
-- the components pass projects `c.debug_skip` and emits the marker as a generated comment.
|
-- This pass projects `c.debug_skip` and emits the marker as a generated comment.
|
||||||
if c.debug_skip then
|
if c.debug_skip then
|
||||||
lines[#lines + 1] = "/* atom_dbg_skip */"
|
lines[#lines + 1] = "/* atom_dbg_skip */"
|
||||||
end
|
end
|
||||||
@@ -536,9 +532,9 @@ local function build_component_lines(c, counts)
|
|||||||
|
|
||||||
local tokens = duffle.split_top_level_commas(c.body)
|
local tokens = duffle.split_top_level_commas(c.body)
|
||||||
for i = 1, #tokens do tokens[i] = duffle.trim(tokens[i]) end
|
for i = 1, #tokens do tokens[i] = duffle.trim(tokens[i]) end
|
||||||
local sig = signature_from_args(c.args)
|
local sig = signature_from_args(c.args)
|
||||||
-- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components).
|
-- Direct lookup against the per-source precomputed `counts` table (built once by count_all_components).
|
||||||
local n = counts[c.name]
|
local n = counts[c.name]
|
||||||
|
|
||||||
if n > 0 then
|
if n > 0 then
|
||||||
emit_macro_body(lines, c, sig, tokens)
|
emit_macro_body(lines, c, sig, tokens)
|
||||||
@@ -557,9 +553,15 @@ end
|
|||||||
|
|
||||||
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
--- Build the boilerplate header lines (the `#ifdef INTELLISENSE_DIRECTIVES` block,
|
||||||
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
--- the `// Auto-generated` comment, the `// Source:` line, and the self-contained `WORD_COUNT` macro definition).
|
||||||
--- @param src SourceFile
|
--- @param dir string -- Absolute source directory
|
||||||
|
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
|
||||||
--- @return string[]
|
--- @return string[]
|
||||||
local function header_boilerplate(src)
|
local function header_boilerplate(dir, sources)
|
||||||
|
local source_lines = { "// Directory: " .. duffle.to_absolute_path(dir) .. "/" }
|
||||||
|
for _, src in ipairs(sources) do
|
||||||
|
source_lines[#source_lines + 1] = "// source: " .. duffle.to_absolute_path(src.path)
|
||||||
|
end
|
||||||
|
local source_blob = table.concat(source_lines, "\n")
|
||||||
return {
|
return {
|
||||||
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
|
-- #pragma once wrapped in #ifdef INTELLISENSE_DIRECTIVES, matching the convention in lottes_tape.h.
|
||||||
-- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling.
|
-- The build does manual unity includes (the user controls include order), so the pragma is only active for IDE/tooling.
|
||||||
@@ -567,7 +569,7 @@ local function header_boilerplate(src)
|
|||||||
"#pragma once",
|
"#pragma once",
|
||||||
"#endif",
|
"#endif",
|
||||||
"// Auto-generated by ps1_meta.lua — DO NOT EDIT",
|
"// Auto-generated by ps1_meta.lua — DO NOT EDIT",
|
||||||
"// Source: " .. duffle.to_absolute_path(src.path),
|
source_blob,
|
||||||
"// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)",
|
"// Component atoms (MipsAtomComp_(ac_*)) -> macro variants (mac_*)",
|
||||||
"",
|
"",
|
||||||
-- Self-contained: define WORD_COUNT if not already defined.
|
-- Self-contained: define WORD_COUNT if not already defined.
|
||||||
@@ -580,30 +582,30 @@ local function header_boilerplate(src)
|
|||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Compute the output path for one source's `.macs.h` file.
|
--- Compute the per-directory output path for `.macs.h`.
|
||||||
--- The pre-rework convention uses the *directory* basename (not the source file basename)
|
--- e.g. any source in `code/duffle/` produces `code/duffle/gen/macs.h` regardless of source filename.
|
||||||
--- e.g. `code/duffle/lottes_tape.h` produces `code/duffle/gen/duffle.macs.h`.
|
--- The directory name is the namespace; the filename does not repeat it.
|
||||||
--- This matches what the C codebase #includes.
|
--- @param dir string -- Absolute source directory
|
||||||
--- @param src SourceFile
|
--- @return string -- Output directory
|
||||||
--- @return string -- the output directory
|
--- @return string -- Full output path
|
||||||
--- @return string -- the full output path
|
local function compute_macs_h_path(dir)
|
||||||
local function compute_macs_h_path(src)
|
local out_dir = dir .. "/" .. GEN_SUBDIR
|
||||||
local out_dir = src.dir .. "/" .. GEN_SUBDIR
|
local out_path = out_dir .. "/" .. MACS_FILENAME
|
||||||
local out_path = out_dir .. "/" .. duffle.basename_no_ext(src.dir) .. ".macs.h"
|
|
||||||
return out_dir, out_path
|
return out_dir, out_path
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Emit a per-source `.macs.h` header with the `mac_X` macros + `WORD_COUNT` entries.
|
--- Emit a per-directory `.macs.h` header with the aggregated `mac_X` macros + `WORD_COUNT` entries.
|
||||||
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
|
--- Writes in BINARY mode so LF line endings are preserved (the git blob is LF; Windows text-mode would emit CRLF and break the byte-identical diff).
|
||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @param src SourceFile
|
--- @param dir string -- Absolute source directory
|
||||||
--- @param components Component[]
|
--- @param sources SourceFile[] -- Sources contributing to this directory (for the header comment)
|
||||||
--- @param counts table<string, integer> -- precomputed word counts (from count_all_components)
|
--- @param components Component[] -- Aggregated components from all sources in this directory
|
||||||
--- @return string|nil -- path to the written file (nil if no components)
|
--- @param counts table<string, integer> -- Precomputed word counts (from count_all_components)
|
||||||
local function emit_component_macros_h(ctx, src, components, counts)
|
--- @return string|nil -- Path to the written file (nil if no components)
|
||||||
|
local function emit_component_macros_h(ctx, dir, sources, components, counts)
|
||||||
if #components == 0 then return nil end
|
if #components == 0 then return nil end
|
||||||
local out_dir, out_path = compute_macs_h_path(src)
|
local out_dir, out_path = compute_macs_h_path(dir)
|
||||||
local lines = header_boilerplate(src)
|
local lines = header_boilerplate(dir, sources)
|
||||||
|
|
||||||
for _, c in ipairs(components) do
|
for _, c in ipairs(components) do
|
||||||
for _, l in ipairs(build_component_lines(c, counts)) do
|
for _, l in ipairs(build_component_lines(c, counts)) do
|
||||||
@@ -638,11 +640,11 @@ local function update_canonical_word_counts(corpus, components, counts)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- @class ComponentDef
|
--- @class ComponentDef
|
||||||
--- @field name string -- bare name (without ac_/mac_ prefix)
|
--- @field name string -- Bare name (without ac_/mac_ prefix)
|
||||||
--- @field line integer -- definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
|
--- @field line integer -- Definition source line (line of `MipsAtomComp_(ac_X)` / `MipsAtomComp_Proc_(ac_X, ...)`)
|
||||||
--- @field path string -- absolute source path of the definition
|
--- @field path string -- Absolute source path of the definition
|
||||||
--- @field kind string -- "comp_bare" | "comp_proc"
|
--- @field kind string -- "comp_bare" | "comp_proc" (atom_proc is NOT a component)
|
||||||
--- @field debug_skip boolean -- mirror of the scanner-owned `a.debug_skip`; consumers read this directly
|
--- @field debug_skip boolean -- Mirror of the scanner-owned `a.debug_skip`; consumers read this directly
|
||||||
|
|
||||||
--- (internal) Populate `corpus.components` with this source's components-by-name map.
|
--- (internal) Populate `corpus.components` with this source's components-by-name map.
|
||||||
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
|
--- First declaration wins; later declarations of the same bare name are dropped and recorded as a collision via `corpus.collisions` (kind = "component").
|
||||||
@@ -675,7 +677,7 @@ local function update_canonical_components(corpus, src, components, metadata)
|
|||||||
-- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
|
-- Identical-shape declarations (same path + line) reuse the first-wins entry without a collision record.
|
||||||
local existing = corpus.components[c.name]
|
local existing = corpus.components[c.name]
|
||||||
if existing.path ~= rel_path or existing.line ~= c.line then
|
if existing.path ~= rel_path or existing.line ~= c.line then
|
||||||
local kind = c.kind or "comp_bare"
|
local kind = c.kind or "comp_bare"
|
||||||
local first_kind = existing.kind or "comp_bare"
|
local first_kind = existing.kind or "comp_bare"
|
||||||
corpus.collisions[#corpus.collisions + 1] = {
|
corpus.collisions[#corpus.collisions + 1] = {
|
||||||
kind = "component",
|
kind = "component",
|
||||||
@@ -740,24 +742,39 @@ function M.run(ctx)
|
|||||||
-- * `corpus.component_body_index[name]` — body / line_of / source index
|
-- * `corpus.component_body_index[name]` — body / line_of / source index
|
||||||
-- The pass writes to the corpus only; consumers read from the corpus directly.
|
-- The pass writes to the corpus only; consumers read from the corpus directly.
|
||||||
|
|
||||||
for _, src in ipairs(corpus.source_order) do
|
-- Per-directory aggregation: every source in the same directory contributes to one `gen/macs.h`.
|
||||||
-- project_components reads from src.scan + does backward lookups on src.text
|
-- The directory itself is the namespace. `corpus.sources_by_dir` preserves source-order within each bucket (matches `corpus.source_order`).
|
||||||
local components = project_components(src.text, src.scan)
|
local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order)
|
||||||
if #components > 0 then
|
for dir, sources in pairs(sources_by_dir) do
|
||||||
-- Compute all component word counts once per source.
|
-- Aggregate components from every source in this directory.
|
||||||
-- Use `corpus.word_counts` so the recursive lookup sees both authored-metadata entries
|
-- `project_components` returns nil for sources with no `MipsAtomComp_` declarations; we skip those.
|
||||||
-- (loaded by word_count_eval.run) AND same-source component entries (populated earlier in this loop by `update_canonical_word_counts`).
|
local aggregated_components = {}
|
||||||
local counts = count_all_components(components, corpus.word_counts)
|
local metadata_per_source = {}
|
||||||
-- Derive cycle_cost + gp0_contrib from the original `MipsAtomComp_` body tokens
|
for _, src in ipairs(sources) do
|
||||||
-- (NOT from the generated `mac_*` variants — those are written to disk above).
|
local per_source = project_components(src.text, src.scan) or {}
|
||||||
local metadata = compute_components_metadata(components, duffle.INSTRUCTION_LATENCY)
|
for _, c in ipairs(per_source) do
|
||||||
local macs_path = emit_component_macros_h(ctx, src, components, counts)
|
aggregated_components[#aggregated_components + 1] = c
|
||||||
|
end
|
||||||
|
if #per_source > 0 then
|
||||||
|
metadata_per_source[src] = compute_components_metadata(per_source, duffle.INSTRUCTION_LATENCY)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #aggregated_components > 0 then
|
||||||
|
-- Compute word counts across the aggregated set. `corpus.word_counts` carries the
|
||||||
|
-- same-source + prior-directory entries so the recursive lookup sees both.
|
||||||
|
local counts = count_all_components(aggregated_components, corpus.word_counts)
|
||||||
|
local macs_path = emit_component_macros_h(ctx, dir, sources, aggregated_components, counts)
|
||||||
if macs_path then
|
if macs_path then
|
||||||
outputs[#outputs + 1] = { macs_h = macs_path }
|
outputs[#outputs + 1] = { macs_h = macs_path }
|
||||||
-- Populate the projections AFTER disk emission (so the byte-identical `.macs.h` contract is preserved before any current-count mutation).
|
-- Populate the projections AFTER disk emission (byte-identical `.macs.h` contract).
|
||||||
update_canonical_word_counts(corpus, components, counts)
|
update_canonical_word_counts(corpus, aggregated_components, counts)
|
||||||
update_canonical_components(corpus, src, components, metadata)
|
for _, src in ipairs(sources) do
|
||||||
update_canonical_component_body_index(corpus, src, components, src.scan)
|
local per_source = project_components(src.text, src.scan) or {}
|
||||||
|
if #per_source > 0 then
|
||||||
|
update_canonical_components(corpus, src, per_source, metadata_per_source[src])
|
||||||
|
update_canonical_component_body_index(corpus, src, per_source, src.scan)
|
||||||
|
end
|
||||||
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|||||||
@@ -73,8 +73,8 @@ local DW_RLE_start_length = DWARF5_RNGLISTS.start_length
|
|||||||
|
|
||||||
-- File-index lookup for the existing main line unit (Unit 2).
|
-- File-index lookup for the existing main line unit (Unit 2).
|
||||||
-- Populated at pass start by `init_file_index_lookup(elf_path)` from the runtime ELF (see `elf_dwarf.read_line_unit_file_table`).
|
-- Populated at pass start by `init_file_index_lookup(elf_path)` from the runtime ELF (see `elf_dwarf.read_line_unit_file_table`).
|
||||||
local _file_index_by_basename = nil -- [basename] = 1-based line-table file index
|
local _file_index_by_basename = nil -- [basename] = 1-based line-table file index
|
||||||
local _file_path_by_index = nil -- [1-based index] = full source path (diagnostics / future consumers)
|
local _file_path_by_index = nil -- [1-based index] = full source path (diagnostics / future consumers)
|
||||||
local _default_atom_source_index = nil -- any valid index used in opaque-row fallbacks
|
local _default_atom_source_index = nil -- any valid index used in opaque-row fallbacks
|
||||||
|
|
||||||
-- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31
|
-- RR_<R_Name> debug-visible variables come from the merged register_alias_registry filtered to aliases whose code is a valid MIPS GPR 0..31
|
||||||
@@ -107,8 +107,8 @@ local ABBREV_TYPED_VIEW_POINTER = 0x6E -- 110: DW_TAG_pointer_type no children
|
|||||||
-- DWARF5 §7.7.3 loclist opcodes.
|
-- DWARF5 §7.7.3 loclist opcodes.
|
||||||
local DW_LLE_end_of_list = 0x00
|
local DW_LLE_end_of_list = 0x00
|
||||||
local DW_LLE_start_length = 0x08
|
local DW_LLE_start_length = 0x08
|
||||||
local DW_OP_reg0 = 0x50 -- base reg op; regN = 0x50 + N
|
local DW_OP_reg0 = 0x50 -- base reg op; regN = 0x50 + N
|
||||||
local DW_OP_breg0 = 0x70 -- base breg op; bregN = 0x70 + N (SLEB offset)
|
local DW_OP_breg0 = 0x70 -- base breg op; bregN = 0x70 + N (SLEB offset)
|
||||||
local DW_OP_piece = 0x93
|
local DW_OP_piece = 0x93
|
||||||
local MIPS_LOAD_DELAY_BYTES = 0x08 -- 1 load word + 1 BD-slot word
|
local MIPS_LOAD_DELAY_BYTES = 0x08 -- 1 load word + 1 BD-slot word
|
||||||
|
|
||||||
@@ -184,15 +184,17 @@ end
|
|||||||
--- Resolve an absolute provenance path to the line-unit file index used by the emitting line program.
|
--- Resolve an absolute provenance path to the line-unit file index used by the emitting line program.
|
||||||
--- Normalizes mixed `/` and `\` separators to a basename and looks it up against the runtime-computed file table populated by `init_file_index_lookup`.
|
--- Normalizes mixed `/` and `\` separators to a basename and looks it up against the runtime-computed file table populated by `init_file_index_lookup`.
|
||||||
---
|
---
|
||||||
--- Fails loudly on an unknown provenance basename: adding a new component source file will produce a clear error message naming the missing basename and listing the .debug_line file table contents,
|
--- Returns 0 (the DWARF `set_file(0)` "no file change" sentinel) when the basename is not in the file table.
|
||||||
--- so the user can either confirm the gcc include order, the unity-root, or the `.debug_line` file table contents.
|
--- This is a normal occurrence: the compiler only adds a file to the `.debug_line` file table when the file has line-numbered content (i.e., code).
|
||||||
--- Silent fallback would mask the new-file case by misattributing component rows to an arbitrary source file.
|
--- Files containing only static-array data (e.g. `MipsAtomComp_` declarations in `gp.atom.c`, `psyq.atom.c`, `pad.atom.c` — the OT-tag inserts, etc.) produce no line numbers,
|
||||||
|
--- so gcc omits them from the file table.
|
||||||
|
--- The DWARF emitter then keeps the previous line-program file state instead of pointing at a file that has no entries to walk.
|
||||||
|
--- A stderr warning is emitted per-miss so the user can audit which files the compiler dropped.
|
||||||
--- @param path string -- absolute provenance path (mixed slashes accepted)
|
--- @param path string -- absolute provenance path (mixed slashes accepted)
|
||||||
--- @return integer -- 1-based line-unit file index
|
--- @return integer -- 1-based line-unit file index, or 0 on miss (DWARF no-change sentinel)
|
||||||
local function resolve_provenance_file_index(path)
|
local function resolve_provenance_file_index(path)
|
||||||
if _file_index_by_basename == nil then
|
if _file_index_by_basename == nil then
|
||||||
error("[dwarf_injection] resolve_provenance_file_index called before init_file_index_lookup. "
|
error("[dwarf_injection] resolve_provenance_file_index called before init_file_index_lookup. Is M.run being entered correctly (with --elf)?")
|
||||||
.. "Is M.run being entered correctly (with --elf)?")
|
|
||||||
end
|
end
|
||||||
if path == nil or path == "" then
|
if path == nil or path == "" then
|
||||||
error("[dwarf_injection] resolve_provenance_file_index: empty path")
|
error("[dwarf_injection] resolve_provenance_file_index: empty path")
|
||||||
@@ -201,19 +203,17 @@ local function resolve_provenance_file_index(path)
|
|||||||
local normalized = path:gsub("\\", "/")
|
local normalized = path:gsub("\\", "/")
|
||||||
-- Take the last path component (the basename).
|
-- Take the last path component (the basename).
|
||||||
local basename = normalized:match("([^/]+)$") or normalized
|
local basename = normalized:match("([^/]+)$") or normalized
|
||||||
local idx = _file_index_by_basename[basename]
|
local idx = _file_index_by_basename[basename]
|
||||||
if idx ~= nil then return idx end
|
if idx ~= nil then return idx end
|
||||||
-- Last-resort exact-path match (handles paths that don't reduce to a known basename).
|
-- Last-resort exact-path match (handles paths that don't reduce to a known basename).
|
||||||
for i, p in pairs(_file_path_by_index) do
|
for i, p in pairs(_file_path_by_index) do
|
||||||
if p and p:gsub("\\", "/") == normalized then return i end
|
if p and p:gsub("\\", "/") == normalized then return i end
|
||||||
end
|
end
|
||||||
-- Build an error message listing the known basenames for fast diagnostics.
|
-- File is in the corpus but gcc omitted it from the .debug_line file table (data-only content).
|
||||||
local known = {}
|
-- Return 0 = DWARF `set_file(0)` no-change sentinel so the line program keeps its prior file state.
|
||||||
for k in pairs(_file_index_by_basename) do known[#known + 1] = k end
|
io.stderr:write(string.format("[dwarf_injection] line-table miss: '%s' (basename '%s') not in .debug_line file table; "
|
||||||
table.sort(known)
|
.. "falling back to set_file(0)\n", path, basename))
|
||||||
error(string.format("[dwarf_injection] resolve_provenance_file_index: unknown provenance basename '%s' (from '%s'). "
|
return 0
|
||||||
.. "Known basenames in the .debug_line file table (%d): %s"
|
|
||||||
, basename, path, #known, table.concat(known, ", ")))
|
|
||||||
end
|
end
|
||||||
|
|
||||||
local DW_FORM_addr = 0x01
|
local DW_FORM_addr = 0x01
|
||||||
@@ -228,7 +228,6 @@ local DW_FORM_sec_offset = 0x17 -- 4-byte section-relative offset (into .d
|
|||||||
|
|
||||||
-- DW_OP_reg0 + DW_OP_piece are declared above (lines 114-116) alongside the other DWARF5 §7.7.3 loclist opcodes.
|
-- DW_OP_reg0 + DW_OP_piece are declared above (lines 114-116) alongside the other DWARF5 §7.7.3 loclist opcodes.
|
||||||
|
|
||||||
|
|
||||||
local DW_ATE_unsigned = 0x07 -- DWARF5 §7.8.1: DW_ATE_unsigned (used for U4 base type)
|
local DW_ATE_unsigned = 0x07 -- DWARF5 §7.8.1: DW_ATE_unsigned (used for U4 base type)
|
||||||
|
|
||||||
-- (DW_LANG_Mips_Assembler = 0x8001 was used in the, but we want this CU to look like a C TU so VSCode's Variables pane treats it as code.)
|
-- (DW_LANG_Mips_Assembler = 0x8001 was used in the, but we want this CU to look like a C TU so VSCode's Variables pane treats it as code.)
|
||||||
@@ -448,9 +447,9 @@ end
|
|||||||
--- Statement-state rules:
|
--- Statement-state rules:
|
||||||
--- * A marked whole atom emits one opaque is_stmt=false range row and no nested component rows; its subprogram symbol/range remains available.
|
--- * A marked whole atom emits one opaque is_stmt=false range row and no nested component rows; its subprogram symbol/range remains available.
|
||||||
--- * Per-row policy at every other PC:
|
--- * Per-row policy at every other PC:
|
||||||
--- - Call-site row of any invocation's first word: is_stmt = true (unconditional; `want_call = true`).
|
--- - Call-site row of any invocation's first word: is_stmt = true (unconditional; `want_call = true`).
|
||||||
--- - Body row of any invocation (first or subsequent): is_stmt = not inv.debug_skip (`want_body = not inv.debug_skip`).
|
--- - Body row of any invocation (first or subsequent): is_stmt = not inv.debug_skip (`want_body = not inv.debug_skip`).
|
||||||
--- - RAW word (no containing invocation): is_stmt = true (unconditional).
|
--- - RAW word (no containing invocation): is_stmt = true (unconditional).
|
||||||
--- * The previous per-word `marked_idx` ancestor walk and the GDB 12 zero-instruction-prologue duplicate row at atom entry are DELETED; the new
|
--- * The previous per-word `marked_idx` ancestor walk and the GDB 12 zero-instruction-prologue duplicate row at atom entry are DELETED; the new
|
||||||
--- first-word emission IS the entry statement.
|
--- first-word emission IS the entry statement.
|
||||||
--- * Whole-atom suppression wins over component markers; no nested inversion.
|
--- * Whole-atom suppression wins over component markers; no nested inversion.
|
||||||
@@ -612,10 +611,8 @@ local function build_atom_sequence(atom)
|
|||||||
-- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call).
|
-- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call).
|
||||||
for ai, anc in ipairs(entry_1_ancestry) do
|
for ai, anc in ipairs(entry_1_ancestry) do
|
||||||
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
|
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
|
||||||
assert(anc.body_lines[1] ~= nil
|
assert(anc.body_lines[1] ~= nil, "dwarf_injection: body_lines[1] missing on first-word entry for inv=" .. tostring(anc.component_name))
|
||||||
, "dwarf_injection: body_lines[1] missing on first-word entry for inv=" .. tostring(anc.component_name))
|
assert(anc.call_path and anc.call_path ~= "", "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
|
||||||
assert(anc.call_path and anc.call_path ~= ""
|
|
||||||
, "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
|
|
||||||
emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
|
emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
|
||||||
local is_outermost = (ai == 1)
|
local is_outermost = (ai == 1)
|
||||||
if not (is_outermost and anc.debug_skip) then
|
if not (is_outermost and anc.debug_skip) then
|
||||||
@@ -640,8 +637,7 @@ local function build_atom_sequence(atom)
|
|||||||
-- all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
|
-- all OTHER ancestors emit body_lines[1] with is_stmt = not debug_skip.
|
||||||
--
|
--
|
||||||
-- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC
|
-- This re-emits the outer ancestor's call-site + body rows at the inner's first word PC
|
||||||
-- for debugger context: source-level stepping now shows the outer body line
|
-- for debugger context: source-level stepping now shows the outer body line (not the inner body line) when stepping into the inner. PROBLEM B fix.
|
||||||
-- (not the inner body line) when stepping into the inner. PROBLEM B fix.
|
|
||||||
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
|
-- The body_lines[1] row references body_first_line_of[anc.id] (= the body's first content line in the parent's source),
|
||||||
-- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call:
|
-- NOT anc.body_lines[1] (= the line of the first WORD, which is wrong when the outer's body starts with a nested call:
|
||||||
-- gdb 12.1 picks the displayed line as the LAST row at the same PC in byte-stream order,
|
-- gdb 12.1 picks the displayed line as the LAST row at the same PC in byte-stream order,
|
||||||
@@ -649,12 +645,9 @@ local function build_atom_sequence(atom)
|
|||||||
local ancestry = ancestry_idx[idx]
|
local ancestry = ancestry_idx[idx]
|
||||||
for ai, anc in ipairs(ancestry) do
|
for ai, anc in ipairs(ancestry) do
|
||||||
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
|
assert(anc.body_lines, "missing body_lines: emitter did not run emission-model")
|
||||||
assert(anc.body_lines[1] ~= nil
|
assert(anc.body_lines[1] ~= nil, string.format("missing body_lines[1] for inv=%s start_pos=%d len=%d", anc.component_name, anc.start_pos, #(anc.body_lines or {})))
|
||||||
, string.format("missing body_lines[1] for inv=%s start_pos=%d len=%d",
|
assert(anc.call_path and anc.call_path ~= "", "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
|
||||||
anc.component_name, anc.start_pos, #(anc.body_lines or {})))
|
emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
|
||||||
assert(anc.call_path and anc.call_path ~= ""
|
|
||||||
, "dwarf_injection: inv.call_path is missing on invocation " .. tostring(anc.component_name) .. "; emitter did not run emission-model.")
|
|
||||||
emit_row(resolve_provenance_file_index(anc.call_path), anc.call_line, true)
|
|
||||||
local is_outermost = (ai == 1)
|
local is_outermost = (ai == 1)
|
||||||
if not (is_outermost and anc.debug_skip) then
|
if not (is_outermost and anc.debug_skip) then
|
||||||
emit_row(resolve_provenance_file_index(anc.def_path), body_first_line_of[anc.id] or anc.body_lines[1], not anc.debug_skip)
|
emit_row(resolve_provenance_file_index(anc.def_path), body_first_line_of[anc.id] or anc.body_lines[1], not anc.debug_skip)
|
||||||
@@ -669,10 +662,8 @@ local function build_atom_sequence(atom)
|
|||||||
-- Marked invocations emit non-statement body rows at every body word; unmarked invocations emit statement body rows.
|
-- Marked invocations emit non-statement body rows at every body word; unmarked invocations emit statement body rows.
|
||||||
assert(inv.body_lines, "missing body_lines: emitter did not run emission-model")
|
assert(inv.body_lines, "missing body_lines: emitter did not run emission-model")
|
||||||
local words_into = idx - inv.start_pos
|
local words_into = idx - inv.start_pos
|
||||||
assert(inv.body_lines[words_into] ~= nil
|
assert(inv.body_lines[words_into] ~= nil, string.format("missing body_lines[%d] for inv=%s start_pos=%d len=%d idx=%d", words_into, inv.component_name, inv.start_pos, #(inv.body_lines or {}), idx))
|
||||||
, string.format("missing body_lines[%d] for inv=%s start_pos=%d len=%d idx=%d",
|
emit_row(resolve_provenance_file_index(inv.def_path), inv.body_lines[words_into], not inv.debug_skip)
|
||||||
words_into, inv.component_name, inv.start_pos, #(inv.body_lines or {}), idx))
|
|
||||||
emit_row(resolve_provenance_file_index(inv.def_path), inv.body_lines[words_into], not inv.debug_skip)
|
|
||||||
else
|
else
|
||||||
-- RAW word: single call-site row, always a statement target (the word itself is unmarked).
|
-- RAW word: single call-site row, always a statement target (the word itself is unmarked).
|
||||||
emit_row(call_file_idx, entry.line, true)
|
emit_row(call_file_idx, entry.line, true)
|
||||||
@@ -712,9 +703,9 @@ end
|
|||||||
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
|
--- `{comp_name, call_file, call_line, comp_file, comp_line, start_pos, end_pos, body_lines, debug_skip}`. `body_lines[k]`
|
||||||
--- is the k-th word's source line within the component body.
|
--- is the k-th word's source line within the component body.
|
||||||
---
|
---
|
||||||
--- @param corpus table -- the corpus from `ctx.shared.corpus`
|
--- @param corpus table -- From `ctx.shared.corpus`
|
||||||
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
|
--- @param addrs table -- ELF symbols keyed by atom name from `elf_dwarf.read_nm`
|
||||||
--- @return table[] -- list of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
|
--- @return table[] -- List of {name, addr, size_bytes, words, entries, invocations, debug_skip?}
|
||||||
local function build_atom_table(corpus, addrs)
|
local function build_atom_table(corpus, addrs)
|
||||||
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
|
-- Cross-ref: keep only atoms present in BOTH the nm symbol table AND `corpus.atoms_by_name`. Output is sorted by ascending addr.
|
||||||
local atoms_by_name = corpus.atoms_by_name or {}
|
local atoms_by_name = corpus.atoms_by_name or {}
|
||||||
@@ -730,8 +721,8 @@ local function build_atom_table(corpus, addrs)
|
|||||||
local word_events = paths.word_events or {}
|
local word_events = paths.word_events or {}
|
||||||
local invocations_proj = paths.invocations or {}
|
local invocations_proj = paths.invocations or {}
|
||||||
-- Build the dense entries list from `word_events`.
|
-- Build the dense entries list from `word_events`.
|
||||||
-- `word_events[i].i` = the 0-based `.word` position
|
-- `word_events[i].i` = the 0-based `.word` position
|
||||||
-- `call_line` = the root atom's physical source line for that word (stamped by emission_model)
|
-- `call_line` = the root atom's physical source line for that word (stamped by emission_model)
|
||||||
local entries = {}
|
local entries = {}
|
||||||
for idx, ev in ipairs(word_events) do
|
for idx, ev in ipairs(word_events) do
|
||||||
entries[#entries + 1] = {
|
entries[#entries + 1] = {
|
||||||
@@ -773,9 +764,8 @@ local function build_atom_table(corpus, addrs)
|
|||||||
|
|
||||||
local out = {}
|
local out = {}
|
||||||
-- Walk every source's atom list (which preserves source order + per-source src_path).
|
-- Walk every source's atom list (which preserves source order + per-source src_path).
|
||||||
-- Cross-ref with the nm symbol table; atoms absent from `addrs` are skipped (an atom
|
-- Cross-ref with the nm symbol table; atoms absent from `addrs` are skipped
|
||||||
-- declared in source but not emitted as a symbol is a metaprogram or atom-info bug, not
|
-- (an atom declared in source but not emitted as a symbol is a metaprogram or atom-info bug, not a source-correlation bug — emit_no_emit would catch it upstream).
|
||||||
-- a source-correlation bug — emit_no_emit would catch it upstream).
|
|
||||||
for _, src in ipairs((corpus and corpus.source_order) or {}) do
|
for _, src in ipairs((corpus and corpus.source_order) or {}) do
|
||||||
local src_path = src.path or ""
|
local src_path = src.path or ""
|
||||||
for _, atom_rec in ipairs(((src.scan or {}).atoms) or {}) do
|
for _, atom_rec in ipairs(((src.scan or {}).atoms) or {}) do
|
||||||
@@ -843,12 +833,11 @@ end
|
|||||||
--- (no `atom_reg` opt-in) are silently skipped — the resulting rbind record will be incomplete and the atom will fail to bind a usable piece chain.
|
--- (no `atom_reg` opt-in) are silently skipped — the resulting rbind record will be incomplete and the atom will fail to bind a usable piece chain.
|
||||||
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
|
--- This is intentional: silently falling back to a hardcoded GPR would mask the missing opt-in.
|
||||||
---
|
---
|
||||||
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level
|
--- Pre-tokenized: `body_tokens` is the scan-source pass's pre-split list of top-level statements (each entry is a single `load_*` call or other statement).
|
||||||
--- statements (each entry is a single `load_*` call or other statement).
|
--- @param body_tokens table[] -- The atom's pre-tokenized body statements (from atom.body_tokens)
|
||||||
--- @param body_tokens table[] -- the atom's pre-tokenized body statements (from atom.body_tokens)
|
--- @param binds_name string -- Expected Binds_X name (skip pairs with mismatching binds)
|
||||||
--- @param binds_name string -- expected Binds_X name (skip pairs with mismatching binds)
|
--- @param registries table -- Merged registries from collect_per_source_registries
|
||||||
--- @param registries table -- merged registries from collect_per_source_registries
|
--- @return table[] -- List of {reg = <MIPS index>, field = <field name>}
|
||||||
--- @return table[] -- list of {reg = <MIPS index>, field = <field name>}
|
|
||||||
local function parse_body_load_pairs(body_tokens, binds_name, registries)
|
local function parse_body_load_pairs(body_tokens, binds_name, registries)
|
||||||
local pairs = {}
|
local pairs = {}
|
||||||
local reg_index_by_name = (registries and registries.register_alias_registry) or {}
|
local reg_index_by_name = (registries and registries.register_alias_registry) or {}
|
||||||
@@ -856,7 +845,7 @@ local function parse_body_load_pairs(body_tokens, binds_name, registries)
|
|||||||
-- The captured ident is `kind`; `inner` holds the parens body for arg parsing.
|
-- The captured ident is `kind`; `inner` holds the parens body for arg parsing.
|
||||||
local load_pattern = "^(load_word|load_half|load_half_u|load_byte|load_byte_u|gte_lw|gte_lwc2)%s*%((.*)%)$"
|
local load_pattern = "^(load_word|load_half|load_half_u|load_byte|load_byte_u|gte_lw|gte_lwc2)%s*%((.*)%)$"
|
||||||
for _, t in ipairs(body_tokens or {}) do
|
for _, t in ipairs(body_tokens or {}) do
|
||||||
local tok = duffle.trim(t.tok or "")
|
local tok = duffle.trim(t.tok or "")
|
||||||
local kind, inner = tok:match(load_pattern)
|
local kind, inner = tok:match(load_pattern)
|
||||||
if kind then
|
if kind then
|
||||||
local args = duffle.split_top_level_commas(inner)
|
local args = duffle.split_top_level_commas(inner)
|
||||||
@@ -882,9 +871,7 @@ local function parse_body_load_pairs(body_tokens, binds_name, registries)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Collect every rbind atom + the matching Binds_X struct + (reg, field) pairs.
|
--- Collect every rbind atom + the matching Binds_X struct + (reg, field) pairs.
|
||||||
---
|
|
||||||
--- Inputs come from the dep-closed `scan-source` pass (the per-source `src.scan` payload is preserved on each `corpus.source_order` entry).
|
--- Inputs come from the dep-closed `scan-source` pass (the per-source `src.scan` payload is preserved on each `corpus.source_order` entry).
|
||||||
---
|
|
||||||
--- Returns:
|
--- Returns:
|
||||||
--- rbind_atoms = {[atom_name] = {binds, fields, regs, byte_size, info_line}}
|
--- rbind_atoms = {[atom_name] = {binds, fields, regs, byte_size, info_line}}
|
||||||
--- rbind_structs = {[binds_name] = {byte_size, fields, atom_names}}
|
--- rbind_structs = {[binds_name] = {byte_size, fields, atom_names}}
|
||||||
@@ -893,9 +880,9 @@ end
|
|||||||
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
|
--- The piece chain uses (DW_OP_regN, DW_OP_piece, ULEB128(field_size)).
|
||||||
---
|
---
|
||||||
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
|
--- Binds fields come from `scan.binds`; the per-source `scan.binds[i].fields` already carries the typed-field record after the scan-source generalization.
|
||||||
--- @param corpus table -- the corpus from `ctx.shared.corpus`
|
--- @param corpus table -- From `ctx.shared.corpus`
|
||||||
--- @param atom_table table[] -- the cross-ref'd atom table from build_atom_table
|
--- @param atom_table table[] -- Cross-ref'd atom table from build_atom_table
|
||||||
--- @param registries table -- merged registries from collect_per_source_registries
|
--- @param registries table -- Merged registries from collect_per_source_registries
|
||||||
--- @return table, table -- (rbind_atoms, rbind_structs)
|
--- @return table, table -- (rbind_atoms, rbind_structs)
|
||||||
local function parse_rbind_atoms(corpus, atom_table, registries)
|
local function parse_rbind_atoms(corpus, atom_table, registries)
|
||||||
registries = registries or {}
|
registries = registries or {}
|
||||||
@@ -929,7 +916,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
|
|||||||
local body_tokens_by_atom = {}
|
local body_tokens_by_atom = {}
|
||||||
for _, src in ipairs((corpus and corpus.source_order) or {}) do
|
for _, src in ipairs((corpus and corpus.source_order) or {}) do
|
||||||
local scan = src.scan
|
local scan = src.scan
|
||||||
if scan then
|
if scan then
|
||||||
for _, atom in ipairs(scan.atoms or {}) do
|
for _, atom in ipairs(scan.atoms or {}) do
|
||||||
body_tokens_by_atom[atom.name] = atom.body_tokens
|
body_tokens_by_atom[atom.name] = atom.body_tokens
|
||||||
end
|
end
|
||||||
@@ -948,8 +935,8 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
|
|||||||
|
|
||||||
for atom_name, ai in pairs(ai_by_atom) do
|
for atom_name, ai in pairs(ai_by_atom) do
|
||||||
if ai.binds then
|
if ai.binds then
|
||||||
local struct = rbind_structs[ai.binds]
|
local struct = rbind_structs[ai.binds]
|
||||||
local body_toks = body_tokens_by_atom[atom_name]
|
local body_toks = body_tokens_by_atom[atom_name]
|
||||||
if struct and body_toks then
|
if struct and body_toks then
|
||||||
local pairs = parse_body_load_pairs(body_toks, ai.binds, registries)
|
local pairs = parse_body_load_pairs(body_toks, ai.binds, registries)
|
||||||
if #pairs > 0 then
|
if #pairs > 0 then
|
||||||
@@ -957,7 +944,7 @@ local function parse_rbind_atoms(corpus, atom_table, registries)
|
|||||||
binds = ai.binds,
|
binds = ai.binds,
|
||||||
fields = struct.fields, -- {name, offset} from scan.binds
|
fields = struct.fields, -- {name, offset} from scan.binds
|
||||||
bytes = struct.bytes,
|
bytes = struct.bytes,
|
||||||
regs = pairs, -- ordered list of {reg, field}
|
regs = pairs, -- Ordered list of {reg, field}
|
||||||
info_line = ai.info_line,
|
info_line = ai.info_line,
|
||||||
}
|
}
|
||||||
table.insert(struct.atom_names, atom_name)
|
table.insert(struct.atom_names, atom_name)
|
||||||
@@ -984,7 +971,8 @@ end
|
|||||||
--- (the final unit, referenced by the main CU's DW_AT_stmt_list).
|
--- (the final unit, referenced by the main CU's DW_AT_stmt_list).
|
||||||
---
|
---
|
||||||
--- This builder extends the main compilation unit.
|
--- This builder extends the main compilation unit.
|
||||||
--- A detached synthetic line unit has no DW_AT_stmt_list referencing it, so gdb ignored it (a previous experiment); byte 13 is the first special opcode, not the extended-opcode marker.
|
--- A detached synthetic line unit has no DW_AT_stmt_list referencing it, so gdb ignored it (a previous experiment);
|
||||||
|
--- byte 13 is the first special opcode, not the extended-opcode marker.
|
||||||
--- The existing final unit already contains hello_gte_tape.c as file index 11 and ends with a valid end_sequence.
|
--- The existing final unit already contains hello_gte_tape.c as file index 11 and ends with a valid end_sequence.
|
||||||
--- We preserve its bytes, append independent atom sequences, and increase only that unit's DWARF32 unit_length.
|
--- We preserve its bytes, append independent atom sequences, and increase only that unit's DWARF32 unit_length.
|
||||||
--- @param existing string -- existing section bytes, byte-for-byte
|
--- @param existing string -- existing section bytes, byte-for-byte
|
||||||
@@ -995,9 +983,7 @@ local function build_dwarf_line_section(existing, atom_table)
|
|||||||
|
|
||||||
-- Build the sequences.
|
-- Build the sequences.
|
||||||
local sequences = {}
|
local sequences = {}
|
||||||
for _, atom in ipairs(atom_table) do
|
for _, atom in ipairs(atom_table) do sequences[#sequences + 1] = build_atom_sequence(atom) end
|
||||||
sequences[#sequences + 1] = build_atom_sequence(atom)
|
|
||||||
end
|
|
||||||
local appended = table.concat(sequences)
|
local appended = table.concat(sequences)
|
||||||
|
|
||||||
-- Walk DWARF32 line units and retain the final unit's bounds.
|
-- Walk DWARF32 line units and retain the final unit's bounds.
|
||||||
@@ -1005,8 +991,8 @@ local function build_dwarf_line_section(existing, atom_table)
|
|||||||
local unit_pos, last_pos, last_length, last_end = 0, nil, nil, nil
|
local unit_pos, last_pos, last_length, last_end = 0, nil, nil, nil
|
||||||
while unit_pos < #existing do
|
while unit_pos < #existing do
|
||||||
if unit_pos + 4 > #existing then return existing end
|
if unit_pos + 4 > #existing then return existing end
|
||||||
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
|
local unit_length = elf_dwarf.read_u32_le(existing, unit_pos)
|
||||||
if unit_length == elf_dwarf.ELF32.dw_dwarf32_terminator then return existing end
|
if unit_length == elf_dwarf.dw_dwarf32_terminator then return existing end
|
||||||
local unit_end_excl = unit_pos + 4 + unit_length
|
local unit_end_excl = unit_pos + 4 + unit_length
|
||||||
if unit_end_excl > #existing then return existing end
|
if unit_end_excl > #existing then return existing end
|
||||||
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
|
last_pos, last_length, last_end = unit_pos, unit_length, unit_end_excl
|
||||||
@@ -1050,13 +1036,13 @@ local function build_dwarf_aranges_section(existing, atom_table)
|
|||||||
-- We bump the unit's length field accordingly.
|
-- We bump the unit's length field accordingly.
|
||||||
--
|
--
|
||||||
-- Unit structure (DWARF4 §7.21):
|
-- Unit structure (DWARF4 §7.21):
|
||||||
-- unit_length (4)
|
-- unit_length (4)
|
||||||
-- version (2)
|
-- version (2)
|
||||||
-- debug_info_offset (4) -- CU DIE offset in .debug_info
|
-- debug_info_offset (4) -- CU DIE offset in .debug_info
|
||||||
-- address_size (1)
|
-- address_size (1)
|
||||||
-- segment_size (1)
|
-- segment_size (1)
|
||||||
-- entries... (4-byte addr + 4-byte length)
|
-- entries... (4-byte addr + 4-byte length)
|
||||||
-- terminator (8 bytes: addr=0, length=0)
|
-- terminator (8 bytes: addr=0, length=0)
|
||||||
|
|
||||||
-- Walk all units and emit each one (preserving existing structure).
|
-- Walk all units and emit each one (preserving existing structure).
|
||||||
-- For the LAST unit, replace the terminator with my entries + new term.
|
-- For the LAST unit, replace the terminator with my entries + new term.
|
||||||
@@ -1067,7 +1053,7 @@ local function build_dwarf_aranges_section(existing, atom_table)
|
|||||||
while i < #existing do
|
while i < #existing do
|
||||||
-- Read this unit's length.
|
-- Read this unit's length.
|
||||||
local ul = elf_dwarf.read_u32_le(existing, i)
|
local ul = elf_dwarf.read_u32_le(existing, i)
|
||||||
if ul == elf_dwarf.ELF32.dw_dwarf32_terminator then
|
if ul == elf_dwarf.dw_dwarf32_terminator then
|
||||||
-- DWARF64 marker - not supported.
|
-- DWARF64 marker - not supported.
|
||||||
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
|
io.stderr:write("[dwarf_injection] WARN: .debug_aranges contains a DWARF64 marker (0xFFFFFFFF); the 64-bit extension is not supported by this metaprogram; passing through unchanged\n")
|
||||||
return existing
|
return existing
|
||||||
@@ -1531,13 +1517,12 @@ end
|
|||||||
--- DW_AT_location = piece-chain (DW_FORM_exprloc)
|
--- DW_AT_location = piece-chain (DW_FORM_exprloc)
|
||||||
--- DW_AT_type = ref4 → structure_type DIE
|
--- DW_AT_type = ref4 → structure_type DIE
|
||||||
---
|
---
|
||||||
--- This function does NOT emit the final 0 byte (root terminator). build_debug_info_section splices bytes ahead of the root terminator
|
--- This function does NOT emit the final 0 byte (root terminator).
|
||||||
--- and preserves existing DIE bytes exactly.
|
--- build_debug_info_section splices bytes ahead of the root terminator and preserves existing DIE bytes exactly.
|
||||||
---
|
---
|
||||||
--- ref4 basis: DW_FORM_ref4 is CU-relative (offset from the first byte of the CU header).
|
--- ref4 basis: DW_FORM_ref4 is CU-relative (offset from the first byte of the CU header).
|
||||||
--- Our inserted DIEs live in the main CU, so every ref4 = (target section offset) - main_cu_offset.
|
--- Our inserted DIEs live in the main CU, so every ref4 = (target section offset) - main_cu_offset.
|
||||||
--- Per-die section offsets are tracked via the running `next_offset` cursor (= section offset of the NEXT byte to emit).
|
--- Per-die section offsets are tracked via the running `next_offset` cursor (= section offset of the NEXT byte to emit).
|
||||||
---
|
|
||||||
--- @param main_cu_offset integer -- 0-based section offset of the main CU's unit_length field
|
--- @param main_cu_offset integer -- 0-based section offset of the main CU's unit_length field
|
||||||
--- @param main_cu_end_excl integer -- 0-based section offset of the first byte AFTER the main CU
|
--- @param main_cu_end_excl integer -- 0-based section offset of the first byte AFTER the main CU
|
||||||
--- @param atom_table table[] -- atoms (with atom.rbind set if rbind; atom.invocations set if mac_X(...) calls)
|
--- @param atom_table table[] -- atoms (with atom.rbind set if rbind; atom.invocations set if mac_X(...) calls)
|
||||||
@@ -1657,6 +1642,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
--
|
--
|
||||||
-- The table is small + explicit — the prototype principle treats the typed-view struct layout as data, not derived state.
|
-- The table is small + explicit — the prototype principle treats the typed-view struct layout as data, not derived state.
|
||||||
local STRUCT_MEMBER_TABLE = {
|
local STRUCT_MEMBER_TABLE = {
|
||||||
|
-- TODO(Ed): This hardcoding is brittle...
|
||||||
|
-- TODO(Ed): Better to just have a table for the fundamental types in duffle/dsl.h, we can derive the rest via typedef parsing...
|
||||||
-- 2-element signed short vector (rare; placeholder for future use).
|
-- 2-element signed short vector (rare; placeholder for future use).
|
||||||
V2_S2 = { byte_size = 4, members = {
|
V2_S2 = { byte_size = 4, members = {
|
||||||
{ name = "x", offset = 0, byte_size = 2 },
|
{ name = "x", offset = 0, byte_size = 2 },
|
||||||
@@ -1793,7 +1780,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
|
emit(uleb128(ABBREV_TYPED_VIEW_POINTER)) -- DW_TAG_pointer_type (abbrev 110; NOT 9; void chain target)
|
||||||
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
|
emit(elf_dwarf.write_u32_le(ref4_of(void_chain_offset))) -- 4-byte ref4: points at the void base_type's tag byte
|
||||||
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
|
-- type_chain_offsets["void|1"] is what step (f) of the per-RR_<R_Name> chain looks up.
|
||||||
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type. The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
|
type_chain_offsets["void|1"] = void_chain_offset -- both the base_type offset and the pointer_type are emitted consecutively; the OUTERMOST is the pointer_type.
|
||||||
|
-- The variable's DW_AT_type must reference the pointer_type, not the base_type. Patch below.
|
||||||
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
|
-- Capture the pointer_type's offset (the last-thing-emitted DIE start) and overwrite the lookup.
|
||||||
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
|
-- The pointer_type was emitted as: uleb(9) (1 byte) + 4-byte ref4 = 5 bytes. Its tag byte is at void_chain_offset + 8 (the base_type's 8 bytes: 1 tag + 5 name + 1 byte_size + 1 encoding).
|
||||||
local ptr_void_offset = void_chain_offset + 8
|
local ptr_void_offset = void_chain_offset + 8
|
||||||
@@ -1909,12 +1897,13 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
end
|
end
|
||||||
local atom_view = (registries.atom_views or {})[atom.name]
|
local atom_view = (registries.atom_views or {})[atom.name]
|
||||||
-- Build the atom-name lookup table once (cheap; O(atom_table)) so step (b) and step (d) can resolve rbind_atom names.
|
-- Build the atom-name lookup table once (cheap; O(atom_table)) so step (b) and step (d) can resolve rbind_atom names.
|
||||||
|
-- TODO(Ed): Bad assignment?
|
||||||
local atom_by_name = atom_by_name or (function() local m = {}; for _, a in ipairs(atom_table) do if a.name then m[a.name] = a end end; return m end)()
|
local atom_by_name = atom_by_name or (function() local m = {}; for _, a in ipairs(atom_table) do if a.name then m[a.name] = a end end; return m end)()
|
||||||
-- step (b) inputs: this atom's `atom_ctx(<rbind_atom>)` (resolved from the registries' atom_ctxs)
|
-- step (b) inputs: this atom's `atom_ctx(<rbind_atom>)` (resolved from the registries' atom_ctxs)
|
||||||
local this_ctx = registries.atom_ctxs and registries.atom_ctxs[atom.name]
|
local this_ctx = registries.atom_ctxs and registries.atom_ctxs[atom.name]
|
||||||
if this_ctx and this_ctx.rbind_atom then
|
if this_ctx and this_ctx.rbind_atom then
|
||||||
local rbind = atom_by_name_global[this_ctx.rbind_atom]
|
local rbind = atom_by_name_global[this_ctx.rbind_atom]
|
||||||
if rbind and rbind.rbind and rbind.rbind.fields then
|
if rbind and rbind.rbind and rbind.rbind.fields then
|
||||||
atom_view_ctx_fields = {}
|
atom_view_ctx_fields = {}
|
||||||
for _, f in ipairs(rbind.rbind.fields) do atom_view_ctx_fields[f.name] = f end
|
for _, f in ipairs(rbind.rbind.fields) do atom_view_ctx_fields[f.name] = f end
|
||||||
if rbind.rbind.regs then
|
if rbind.rbind.regs then
|
||||||
@@ -1934,11 +1923,11 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
end
|
end
|
||||||
if my_phase_label then
|
if my_phase_label then
|
||||||
local group = (registries.atom_phases or {})[my_phase_label]
|
local group = (registries.atom_phases or {})[my_phase_label]
|
||||||
if group and group.atoms then
|
if group and group.atoms then
|
||||||
for _, group_atom_name in ipairs(group.atoms) do
|
for _, group_atom_name in ipairs(group.atoms) do
|
||||||
if group_atom_name ~= atom.name then
|
if group_atom_name ~= atom.name then
|
||||||
local cand = atom_by_name_global[group_atom_name]
|
local cand = atom_by_name_global[group_atom_name]
|
||||||
if cand and cand.rbind and cand.rbind.fields then
|
if cand and cand.rbind and cand.rbind.fields then
|
||||||
atom_view_phase_fields = {}
|
atom_view_phase_fields = {}
|
||||||
for _, f in ipairs(cand.rbind.fields) do atom_view_phase_fields[f.name] = f end
|
for _, f in ipairs(cand.rbind.fields) do atom_view_phase_fields[f.name] = f end
|
||||||
if cand.rbind.regs then
|
if cand.rbind.regs then
|
||||||
@@ -1959,7 +1948,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
-- (a) per-atom callsite atom_type(R_X, <T>): most specific; user explicit override for THIS atom only.
|
-- (a) per-atom callsite atom_type(R_X, <T>): most specific; user explicit override for THIS atom only.
|
||||||
function(r_name, alias_code)
|
function(r_name, alias_code)
|
||||||
local override = atom_view and atom_view.reg_type_overrides and atom_view.reg_type_overrides[r_name]
|
local override = atom_view and atom_view.reg_type_overrides and atom_view.reg_type_overrides[r_name]
|
||||||
if override and override.pointer_depth and override.pointer_depth > 0 then
|
if override and override.pointer_depth and override.pointer_depth > 0 then
|
||||||
return type_chain_offsets[override.type_name .. "|" .. override.pointer_depth]
|
return type_chain_offsets[override.type_name .. "|" .. override.pointer_depth]
|
||||||
end
|
end
|
||||||
end,
|
end,
|
||||||
@@ -1967,7 +1956,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
function(r_name, alias_code)
|
function(r_name, alias_code)
|
||||||
local ctx_field_name = reg_to_field_ctx and reg_to_field_ctx[alias_code]
|
local ctx_field_name = reg_to_field_ctx and reg_to_field_ctx[alias_code]
|
||||||
local ctx_f = ctx_field_name and atom_view_ctx_fields and atom_view_ctx_fields[ctx_field_name]
|
local ctx_f = ctx_field_name and atom_view_ctx_fields and atom_view_ctx_fields[ctx_field_name]
|
||||||
if ctx_f and ctx_f.pointer_depth and ctx_f.pointer_depth > 0 then
|
if ctx_f and ctx_f.pointer_depth and ctx_f.pointer_depth > 0 then
|
||||||
return type_chain_offsets[ctx_f.type_name .. "|" .. ctx_f.pointer_depth]
|
return type_chain_offsets[ctx_f.type_name .. "|" .. ctx_f.pointer_depth]
|
||||||
end
|
end
|
||||||
end,
|
end,
|
||||||
@@ -1975,7 +1964,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
function(r_name, alias_code)
|
function(r_name, alias_code)
|
||||||
local field_name = reg_to_field[alias_code]
|
local field_name = reg_to_field[alias_code]
|
||||||
local f = field_name and field_type_by_name[field_name]
|
local f = field_name and field_type_by_name[field_name]
|
||||||
if f and f.pointer_depth and f.pointer_depth > 0 then
|
if f and f.pointer_depth and f.pointer_depth > 0 then
|
||||||
return type_chain_offsets[f.type_name .. "|" .. f.pointer_depth]
|
return type_chain_offsets[f.type_name .. "|" .. f.pointer_depth]
|
||||||
end
|
end
|
||||||
end,
|
end,
|
||||||
@@ -1983,12 +1972,13 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
function(r_name, alias_code)
|
function(r_name, alias_code)
|
||||||
local phase_field_name = reg_to_field_phase and reg_to_field_phase[alias_code]
|
local phase_field_name = reg_to_field_phase and reg_to_field_phase[alias_code]
|
||||||
local phase_f = phase_field_name and atom_view_phase_fields and atom_view_phase_fields[phase_field_name]
|
local phase_f = phase_field_name and atom_view_phase_fields and atom_view_phase_fields[phase_field_name]
|
||||||
if phase_f and phase_f.pointer_depth and phase_f.pointer_depth > 0 then
|
if phase_f and phase_f.pointer_depth and phase_f.pointer_depth > 0 then
|
||||||
return type_chain_offsets[phase_f.type_name .. "|" .. phase_f.pointer_depth]
|
return type_chain_offsets[phase_f.type_name .. "|" .. phase_f.pointer_depth]
|
||||||
end
|
end
|
||||||
end,
|
end,
|
||||||
-- (e) enum-site atom_type(<T>) default on the registry entry.
|
-- (e) enum-site atom_type(<T>) default on the registry entry.
|
||||||
function(r_name, alias_code)
|
function(r_name, alias_code)
|
||||||
|
-- TODO(Ed): Bad definition?
|
||||||
if alias and alias.default_type and alias.default_depth and alias.default_depth > 0 then
|
if alias and alias.default_type and alias.default_depth and alias.default_depth > 0 then
|
||||||
return type_chain_offsets[alias.default_type .. "|" .. alias.default_depth]
|
return type_chain_offsets[alias.default_type .. "|" .. alias.default_depth]
|
||||||
end
|
end
|
||||||
@@ -1997,8 +1987,8 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
|
|
||||||
-- Iterate `by_alias` in sorted order; Lua's pairs() is non-deterministic, so sorting ensures byte-identical DWARF output across builds.
|
-- Iterate `by_alias` in sorted order; Lua's pairs() is non-deterministic, so sorting ensures byte-identical DWARF output across builds.
|
||||||
for _, r_name in ipairs(by_alias_order) do
|
for _, r_name in ipairs(by_alias_order) do
|
||||||
local alias = by_alias[r_name]
|
local alias = by_alias[r_name]
|
||||||
local rr_name = "RR_" .. strip_r_prefix(r_name)
|
local rr_name = "RR_" .. strip_r_prefix(r_name)
|
||||||
local alias_code = alias.code
|
local alias_code = alias.code
|
||||||
emit(uleb128(ABBREV_VARIABLE))
|
emit(uleb128(ABBREV_VARIABLE))
|
||||||
emit(rr_name .. "\0") -- DW_FORM_string (DW_AT_name)
|
emit(rr_name .. "\0") -- DW_FORM_string (DW_AT_name)
|
||||||
@@ -2020,7 +2010,7 @@ local function build_inserted_children(main_cu_offset, main_cu_end_excl, atom_ta
|
|||||||
-- Two PC ranges cover every field: [atom.addr, last_load+8) describes each field as tape memory (DW_OP_bregN + offset) piece,
|
-- Two PC ranges cover every field: [atom.addr, last_load+8) describes each field as tape memory (DW_OP_bregN + offset) piece,
|
||||||
-- and [last_load+8, atom.end) describes each field as a GPR (DW_OP_regN) piece.
|
-- and [last_load+8, atom.end) describes each field as a GPR (DW_OP_regN) piece.
|
||||||
if atom.rbind then
|
if atom.rbind then
|
||||||
local binds_name = atom.rbind.binds
|
local binds_name = atom.rbind.binds
|
||||||
local loclists_offset = loclists_offsets[atom.name] or 0
|
local loclists_offset = loclists_offsets[atom.name] or 0
|
||||||
emit(uleb128(ABBREV_BIND_VAR_LOCLIST))
|
emit(uleb128(ABBREV_BIND_VAR_LOCLIST))
|
||||||
emit("bind_args\0") -- DW_FORM_string (DW_AT_name)
|
emit("bind_args\0") -- DW_FORM_string (DW_AT_name)
|
||||||
@@ -2073,7 +2063,7 @@ end
|
|||||||
---
|
---
|
||||||
--- Fails safely by returning existing sections unchanged if the table walker can't find the table terminator (malformed input).
|
--- Fails safely by returning existing sections unchanged if the table walker can't find the table terminator (malformed input).
|
||||||
---
|
---
|
||||||
--- @param existing string -- existing .debug_abbrev bytes, byte-for-byte
|
--- @param existing string -- existing .debug_abbrev bytes, byte-for-byte
|
||||||
--- @param main_abbrev_offset integer -- 0-based offset into `existing` of the main CU's abbrev table
|
--- @param main_abbrev_offset integer -- 0-based offset into `existing` of the main CU's abbrev table
|
||||||
--- @return string, integer -- (new_abbrev_bytes, offset_where_duplicate_table_starts = #existing)
|
--- @return string, integer -- (new_abbrev_bytes, offset_where_duplicate_table_starts = #existing)
|
||||||
local function build_debug_abbrev_section(existing, main_abbrev_offset)
|
local function build_debug_abbrev_section(existing, main_abbrev_offset)
|
||||||
@@ -2091,7 +2081,7 @@ local function build_debug_abbrev_section(existing, main_abbrev_offset)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Build the new .debug_str: existing strings + new strings appended.
|
--- Build the new .debug_str: existing strings + new strings appended.
|
||||||
--- @param existing string -- existing .debug_str bytes, byte-for-byte
|
--- @param existing string -- existing .debug_str bytes, byte-for-byte
|
||||||
--- @param atom_table table[]
|
--- @param atom_table table[]
|
||||||
--- @param registries table -- merged registries from collect_per_source_registries
|
--- @param registries table -- merged registries from collect_per_source_registries
|
||||||
--- @return string -- existing bytes plus the deterministic appended strings
|
--- @return string -- existing bytes plus the deterministic appended strings
|
||||||
@@ -2101,7 +2091,6 @@ local function build_debug_str_section(existing, atom_table, registries)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Build the new .debug_info: SPLICE inserted DIEs into the MAIN CU as children.
|
--- Build the new .debug_info: SPLICE inserted DIEs into the MAIN CU as children.
|
||||||
---
|
|
||||||
--- This implementation:
|
--- This implementation:
|
||||||
--- 1. Builds the inserted-children bytes (base_type, struct_types, subprograms with their RR_* + bind_args children) via build_inserted_children.
|
--- 1. Builds the inserted-children bytes (base_type, struct_types, subprograms with their RR_* + bind_args children) via build_inserted_children.
|
||||||
--- 2. Patches the main CU's `unit_length` field to account for the inserted bytes.
|
--- 2. Patches the main CU's `unit_length` field to account for the inserted bytes.
|
||||||
@@ -2111,14 +2100,14 @@ end
|
|||||||
---
|
---
|
||||||
--- The crt CU (everything before main_cu_start) is preserved.
|
--- The crt CU (everything before main_cu_start) is preserved.
|
||||||
|
|
||||||
--- @param existing string -- existing .debug_info section bytes
|
--- @param existing string -- existing .debug_info section bytes
|
||||||
--- @param main_cu_start integer -- 0-based offset of the main CU's unit_length field
|
--- @param main_cu_start integer -- 0-based offset of the main CU's unit_length field
|
||||||
--- @param main_cu_end_excl integer -- 0-based offset of the first byte AFTER the main CU
|
--- @param main_cu_end_excl integer -- 0-based offset of the first byte AFTER the main CU
|
||||||
--- @param new_abbrev_offset integer -- 0-based offset into the new .debug_abbrev of the duplicate main table
|
--- @param new_abbrev_offset integer -- 0-based offset into the new .debug_abbrev of the duplicate main table
|
||||||
--- @param atom_table table[]
|
--- @param atom_table table[]
|
||||||
--- @param rbind_structs table -- {[binds_name] = {bytes, fields, atom_names}}
|
--- @param rbind_structs table -- {[binds_name] = {bytes, fields, atom_names}}
|
||||||
--- @param loclists_offsets table -- {[atom_name] = section-relative offset}
|
--- @param loclists_offsets table -- {[atom_name] = section-relative offset}
|
||||||
--- @param registries table -- merged registries from collect_per_source_registries
|
--- @param registries table -- merged registries from collect_per_source_registries
|
||||||
--- @return string -- the rebuilt .debug_info bytes
|
--- @return string -- the rebuilt .debug_info bytes
|
||||||
local function build_debug_info_section(existing, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries)
|
local function build_debug_info_section(existing, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries)
|
||||||
-- 1) Build the inserted children bytes (just before the main CU's root terminator).
|
-- 1) Build the inserted children bytes (just before the main CU's root terminator).
|
||||||
@@ -2176,13 +2165,13 @@ local SECTION_WRITERS = {
|
|||||||
-- Write a list of `{name, data}` section records to disk via SECTION_WRITERS.
|
-- Write a list of `{name, data}` section records to disk via SECTION_WRITERS.
|
||||||
-- @param results table[] -- list of `{name=, data=}` records to write
|
-- @param results table[] -- list of `{name=, data=}` records to write
|
||||||
-- @param ctx PassCtx
|
-- @param ctx PassCtx
|
||||||
-- @param basename string -- output file basename (e.g. "hello_gte")
|
-- @param basename string -- output file basename (e.g. "hello_gte")
|
||||||
-- @return table -- list of {name_bin = path} entries to append to M.run's outputs
|
-- @return table -- list of {name_bin = path} entries to append to M.run's outputs
|
||||||
local function write_sections(results, ctx, basename)
|
local function write_sections(results, ctx, basename)
|
||||||
local outputs = {}
|
local outputs = {}
|
||||||
for _, r in ipairs(results) do
|
for _, r in ipairs(results) do
|
||||||
local path = SECTION_WRITERS[r.name](ctx.out_root, basename)
|
local path = SECTION_WRITERS[r.name](ctx.out_root, basename)
|
||||||
local f = io.open(path, "wb")
|
local f = io.open(path, "wb")
|
||||||
if not f then
|
if not f then
|
||||||
io.stderr:write(string.format("[dwarf_injection] failed to open %s for write\n", path))
|
io.stderr:write(string.format("[dwarf_injection] failed to open %s for write\n", path))
|
||||||
else
|
else
|
||||||
@@ -2209,7 +2198,7 @@ function M.run(ctx)
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- Guard: --elf is required.
|
-- Guard: --elf is required.
|
||||||
local elf_path = ctx.flags and ctx.flags.elf_path
|
local elf_path = ctx.flags and ctx.flags.elf_path
|
||||||
if not elf_path or elf_path == "" then
|
if not elf_path or elf_path == "" then
|
||||||
io.stderr:write("[dwarf_injection] --elf flag missing\n")
|
io.stderr:write("[dwarf_injection] --elf flag missing\n")
|
||||||
return { outputs = {}, errors = {}, warnings = {} }
|
return { outputs = {}, errors = {}, warnings = {} }
|
||||||
@@ -2220,7 +2209,8 @@ function M.run(ctx)
|
|||||||
|
|
||||||
-- Read the existing DWARF sections directly (no subprocess; io.open + manual ELF32 section-header walk).
|
-- Read the existing DWARF sections directly (no subprocess; io.open + manual ELF32 section-header walk).
|
||||||
-- We need all 8 sections: .debug_line / .debug_aranges / .debug_rnglists get extended (additional rows appended to the existing unit),
|
-- We need all 8 sections: .debug_line / .debug_aranges / .debug_rnglists get extended (additional rows appended to the existing unit),
|
||||||
-- and .debug_info / .debug_abbrev / .debug_str / .debug_loc / .debug_loclists get spliced (the main CU's unit_length is patched; no new compile unit is appended; .debug_loc/.debug_loclists may not exist in the source ELF so we add-section them on splice).
|
-- and .debug_info / .debug_abbrev / .debug_str / .debug_loc / .debug_loclists get spliced
|
||||||
|
-- (the main CU's unit_length is patched; no new compile unit is appended; .debug_loc/.debug_loclists may not exist in the source ELF so we add-section them on splice).
|
||||||
-- The per-section dispatch is inlined in the writers loop below.
|
-- The per-section dispatch is inlined in the writers loop below.
|
||||||
local existing_sections = elf_dwarf.read_elf_sections(elf_path, {
|
local existing_sections = elf_dwarf.read_elf_sections(elf_path, {
|
||||||
".debug_line", ".debug_aranges", ".debug_rnglists",
|
".debug_line", ".debug_aranges", ".debug_rnglists",
|
||||||
@@ -2235,12 +2225,11 @@ function M.run(ctx)
|
|||||||
init_file_index_lookup(elf_path)
|
init_file_index_lookup(elf_path)
|
||||||
-- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation).
|
-- Skip state lives in `corpus.atoms_by_name[*].debug_skip` (whole-atom) and `atom.paths.invocations[*].debug_skip` (per-invocation).
|
||||||
-- `corpus` is the sole canonical source projection.
|
-- `corpus` is the sole canonical source projection.
|
||||||
local corpus = (ctx.shared and ctx.shared.corpus) or {}
|
local corpus = (ctx.shared and ctx.shared.corpus) or {}
|
||||||
local registries = collect_per_source_registries(corpus)
|
local registries = collect_per_source_registries(corpus)
|
||||||
-- Read nm symbols (the ONLY disk-side input to the atom table) and join
|
-- Read nm symbols (the ONLY disk-side input to the atom table) and join them against `corpus.atoms_by_name` + `atom.paths` for word rows + invocation ancestry.
|
||||||
-- them against `corpus.atoms_by_name` + `atom.paths` for word rows + invocation ancestry.
|
|
||||||
-- Disk source-map/provenance text is not consulted (those are diagnostic artifacts; semantic inputs are in memory).
|
-- Disk source-map/provenance text is not consulted (those are diagnostic artifacts; semantic inputs are in memory).
|
||||||
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
|
local addrs = elf_dwarf.read_nm(ctx.flags.elf_path)
|
||||||
local atom_table = build_atom_table(corpus, addrs)
|
local atom_table = build_atom_table(corpus, addrs)
|
||||||
|
|
||||||
-- Detect rbind atoms + index Binds_* struct fields from the corpus.
|
-- Detect rbind atoms + index Binds_* struct fields from the corpus.
|
||||||
@@ -2264,7 +2253,7 @@ function M.run(ctx)
|
|||||||
duffle.ensure_dir(ctx.out_root)
|
duffle.ensure_dir(ctx.out_root)
|
||||||
|
|
||||||
-- Step 0: layout validation. Bail out safely if the .debug_info layout doesn't match what we expect (crt CU + DWARF5 main CU + final 0 byte).
|
-- Step 0: layout validation. Bail out safely if the .debug_info layout doesn't match what we expect (crt CU + DWARF5 main CU + final 0 byte).
|
||||||
-- A layout mismatch means the gcc emission changed; the safest response is to leave existing sections unchanged and emit no synthetic data, so the build's debug-info step never silently produces broken DWARF.
|
-- A layout mismatch means the gcc emission changed; the safest response is to leave existing sections unchanged and emit no synthetic data, so the build's debug-info step never silently produces broken DWARF.
|
||||||
local existing_info = existing_sections[".debug_info"] or ""
|
local existing_info = existing_sections[".debug_info"] or ""
|
||||||
local existing_abbrev = existing_sections[".debug_abbrev"] or ""
|
local existing_abbrev = existing_sections[".debug_abbrev"] or ""
|
||||||
local main_cu_start, main_cu_end_excl, main_abbrev_offset = find_main_cu_layout(existing_info)
|
local main_cu_start, main_cu_end_excl, main_abbrev_offset = find_main_cu_layout(existing_info)
|
||||||
@@ -2299,7 +2288,7 @@ function M.run(ctx)
|
|||||||
local new_info = build_debug_info_section(existing_info, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries)
|
local new_info = build_debug_info_section(existing_info, main_cu_start, main_cu_end_excl, new_abbrev_offset, atom_table, rbind_structs, loclists_offsets, registries)
|
||||||
|
|
||||||
-- Step 2b: rebuild .debug_str now that we know which RR_<R_Name> entries get emitted.
|
-- Step 2b: rebuild .debug_str now that we know which RR_<R_Name> entries get emitted.
|
||||||
-- This aligns with build_debug_info_section's by_alias loop.
|
-- This aligns with build_debug_info_section's by_alias loop.
|
||||||
local new_str = build_debug_str_section(existing_sections[".debug_str"] or "", atom_table, registries)
|
local new_str = build_debug_str_section(existing_sections[".debug_str"] or "", atom_table, registries)
|
||||||
|
|
||||||
-- Step 3-5: independent sections.
|
-- Step 3-5: independent sections.
|
||||||
|
|||||||
@@ -41,7 +41,6 @@ local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
|||||||
-- Convert the recursive walk's body-relative line numbers into physical source lines once.
|
-- Convert the recursive walk's body-relative line numbers into physical source lines once.
|
||||||
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
|
-- The walker builds `line_of` from `body_text` and stamps body-relative line numbers (1..N) into `item.line` and `invocation.call_line`.
|
||||||
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
|
-- This function converts those values to physical source lines at the close site with the forwarded source `line_of` closure.
|
||||||
--
|
|
||||||
-- `call_line` discipline:
|
-- `call_line` discipline:
|
||||||
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
|
-- * ROOT invocations (`inv.parent_id == 0`) receive body-relative `call_line` values directly from `M.LineIndex(body_text)` in the walker.
|
||||||
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
|
-- The source `line_of` closure supplies physical lines at the close site, so this function converts each root value exactly once.
|
||||||
@@ -59,10 +58,9 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
|
|||||||
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
|
-- `root_body_line` is the physical source line of the ATOM HEADER byte containing the opening `{`; that byte is one byte BEFORE `atom_record.body_off`.
|
||||||
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
|
-- The walker assigns line 2 to the body's first content line because line 1 is the trailing `\n` after `{`. Body-text line k therefore maps to `root_body_line + (k - 1)`.
|
||||||
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
|
-- `body_off - 1` points at the opening `{`, whose line index identifies the header line. `body_off` points after `{` and would shift every word row forward by one line.
|
||||||
local root_body_line = root_line_of(atom_record.body_off - 1)
|
local root_body_line = root_line_of(atom_record.body_off - 1) or atom_record.line or 0
|
||||||
or atom_record.line or 0
|
|
||||||
local component_index = corpus.component_body_index or {}
|
local component_index = corpus.component_body_index or {}
|
||||||
local word_items = {}
|
local word_items = {}
|
||||||
|
|
||||||
for _, item in ipairs(projection.items) do
|
for _, item in ipairs(projection.items) do
|
||||||
if item.kind == "word" then word_items[#word_items + 1] = item end
|
if item.kind == "word" then word_items[#word_items + 1] = item end
|
||||||
@@ -102,7 +100,7 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- Normalize `inv.call_line` to a physical source line.
|
-- Normalize `inv.call_line` to a physical source line.
|
||||||
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
|
-- * ROOT invocations (`parent_id == 0`) carry body-relative `call_line` values from `M.LineIndex(body_text)`; convert them once with `root_body_line`.
|
||||||
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
|
-- * INNER invocations (`parent_id ~= 0`) carry physical `call_line` values from the component's `line_of`; retain them unchanged.
|
||||||
for _, inv in ipairs(projection.invocations) do
|
for _, inv in ipairs(projection.invocations) do
|
||||||
if inv.parent_id == 0 then
|
if inv.parent_id == 0 then
|
||||||
@@ -114,14 +112,14 @@ local function stamp_root_provenance(projection, atom_record, src, corpus)
|
|||||||
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
|
-- `atoms_source_map` and `dwarf_injection` read `inv.body_lines[k]` directly from the invocation record created here.
|
||||||
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
|
-- Component words already carry physical `item.line` values from the walker's COMPONENT line index, so `body_line_for` returns them unchanged.
|
||||||
for _, inv in ipairs(projection.invocations) do
|
for _, inv in ipairs(projection.invocations) do
|
||||||
local sw = inv.start_word
|
local sw = inv.start_word
|
||||||
local ew = inv.end_word
|
local ew = inv.end_word
|
||||||
local bls = {}
|
local bls = {}
|
||||||
for i = sw, ew do
|
for i = sw, ew do
|
||||||
local it = projection.items and projection.items[i]
|
local it = projection.items and projection.items[i]
|
||||||
if it and it.kind == "word" then
|
if it and it.kind == "word" then
|
||||||
local fake_event = { invocation_ids = { inv.id } }
|
local fake_event = { invocation_ids = { inv.id } }
|
||||||
bls[#bls + 1] = body_line_for(fake_event, it) or 0
|
bls[#bls + 1] = body_line_for(fake_event, it) or 0
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
inv.body_lines = bls
|
inv.body_lines = bls
|
||||||
@@ -190,11 +188,11 @@ function M.run(ctx)
|
|||||||
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
if type(corpus.source_order) ~= "table" then error("emission_model: ctx.shared.corpus.source_order is required", 0) end
|
||||||
|
|
||||||
-- Project once, collect errors + warnings for one atom.
|
-- Project once, collect errors + warnings for one atom.
|
||||||
-- Kind must be one of: atom | raw_atom | comp_bare | comp_proc.
|
-- Kind must be one of: atom | atom_proc | raw_atom | comp_bare | comp_proc.
|
||||||
local function process_atom(atom, src)
|
local function process_atom(atom, src)
|
||||||
if not (atom and atom.body) then return end
|
if not (atom and atom.body) then return end
|
||||||
local kind = atom.kind
|
local kind = atom.kind
|
||||||
if kind ~= "atom" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
if kind ~= "atom" and kind ~= "atom_proc" and kind ~= "raw_atom" and kind ~= "comp_bare" and kind ~= "comp_proc" then
|
||||||
return
|
return
|
||||||
end
|
end
|
||||||
local proj = project_atom(atom, src, corpus)
|
local proj = project_atom(atom, src, corpus)
|
||||||
@@ -217,7 +215,7 @@ function M.run(ctx)
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
-- Walk `corpus.source_order`; within each source, visit atoms followed by raw_atoms.
|
||||||
-- Recognized kinds (atom | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
-- Recognized kinds (atom | atom_proc | raw_atom | comp_bare | comp_proc) each receive the atom.paths projection via duffle.project_emission.
|
||||||
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
-- Components are macros inlined into atom bodies; focused tests and isolated component analyses consume atom.paths directly.
|
||||||
for _, src in ipairs(corpus.source_order) do
|
for _, src in ipairs(corpus.source_order) do
|
||||||
local scan = src.scan or {}
|
local scan = src.scan or {}
|
||||||
|
|||||||
+69
-52
@@ -3,12 +3,19 @@
|
|||||||
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
--- Reads the pre-scanned SourceScan payload (produced once upstream by `duffle.scan_source`)
|
||||||
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
--- for `MipsAtom_(name)` and `MipsCode code_<name>` declarations, computes the word offset
|
||||||
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
--- from each `atom_offset(F, T)` marker to its target `atom_label(T)` declaration, and emits
|
||||||
--- `<dir_basename>.offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
--- `gen/offsets.h` with one `#define _atom_offset_F_T = N` per branch.
|
||||||
|
---
|
||||||
|
--- Per-directory aggregation: every source in the same directory contributes to the same `gen/offsets.h`.
|
||||||
|
--- The directory itself is the namespace; the filename does not repeat the module name.
|
||||||
|
---
|
||||||
|
--- (Task 12.16 note: atom-namespaced enum names — e.g., `atom_offset__normalize_v3s4__srav_path__aligned_done` —
|
||||||
|
--- were considered to prevent cross-atom label collisions, but the C-side `atom_offset(F, T)` macro in
|
||||||
|
--- `code/duffle/dsl.atom.h` doesn't know the current atom_name at expansion time, so any namespacing
|
||||||
|
--- on the metaprogram side breaks the C build. Reverted. The C-side would need a per-atom
|
||||||
|
--- `CURRENT_ATOM` #define (set by `MipsAtom_`/`MipsAtom_Proc_` macros) plus an updated `atom_offset`
|
||||||
|
--- macro that uses it. That's a coordinated refactor — deferred to a future track.)
|
||||||
---
|
---
|
||||||
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
--- The offset is `target_word - branch_word - 1` (the standard MIPS branch-immediate encoding: branch_offset = relative_pc_in_words - 1).
|
||||||
---
|
|
||||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex,
|
|
||||||
--- Lua 5.3 compatible.
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Module-scope requires + package.path setup
|
-- Module-scope requires + package.path setup
|
||||||
@@ -16,12 +23,11 @@
|
|||||||
|
|
||||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||||
-- Uses `debug.getinfo` to find this file's own directory, so it works
|
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||||
-- both standalone and when require'd from the orchestrator.
|
|
||||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Constants
|
-- Constants
|
||||||
@@ -39,35 +45,35 @@ local OFFSET_MACRO_COL = 44
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
--- @field scan table -- pre-scanned SourceScan payload (from duffle.scan_source)
|
--- @field scan table -- Pre-scanned SourceScan payload (from duffle.scan_source)
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field shared table -- cross-pass shared state
|
--- @field shared table -- Cross-pass shared state
|
||||||
--- @field shared.corpus table -- canonical corpus projection
|
--- @field shared.corpus table -- Corpus projection
|
||||||
--- @field shared.word_counts table
|
--- @field shared.word_counts table
|
||||||
--- @field out_root string -- output root (e.g. "build/gen")
|
--- @field out_root string -- Output root (e.g. "build/gen")
|
||||||
|
|
||||||
--- @class PassResult
|
--- @class PassResult
|
||||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||||
|
|
||||||
--- @class BranchOffset
|
--- @class BranchOffset
|
||||||
--- @field tag string -- the marker tag (e.g. "F" in `atom_offset(F, T)`)
|
--- @field tag string -- Marker tag (e.g. "F" in `atom_offset(F, T)`)
|
||||||
--- @field target string -- the target label name (e.g. "T" in `atom_offset(F, T)`)
|
--- @field target string -- Target label name (e.g. "T" in `atom_offset(F, T)`)
|
||||||
--- @field branch_word integer -- branch word position within the atom body
|
--- @field branch_word integer -- Branch word position within the atom body
|
||||||
--- @field offset integer -- computed per consuming instruction (see `compute_offsets`)
|
--- @field offset integer -- Computed per consuming instruction (see `compute_offsets`)
|
||||||
--- @field consuming_encoder string|nil -- the instruction consuming the offset (e.g. "branch_le_zero", "jump", "call_addr")
|
--- @field consuming_encoder string|nil -- Instruction consuming the offset (e.g. "branch_le_zero", "jump", "call_addr")
|
||||||
--- @field consuming_arg_pos integer|nil -- 1-based arg position within the consuming instruction's arg list
|
--- @field consuming_arg_pos integer|nil -- 1-based arg position within the consuming instruction's arg list
|
||||||
|
|
||||||
--- @class AtomData
|
--- @class AtomData
|
||||||
--- @field name string -- atom name
|
--- @field name string -- Atom name
|
||||||
--- @field total_words integer -- total word count of the atom body
|
--- @field total_words integer -- Total word count of the atom body
|
||||||
--- @field offsets BranchOffset[] -- per-branch offset list
|
--- @field offsets BranchOffset[] -- Per-branch offset list
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Canonical marker projection
|
-- Canonical marker projection
|
||||||
@@ -99,7 +105,7 @@ local function project_markers(markers)
|
|||||||
local state = { labels = {}, branches = {} }
|
local state = { labels = {}, branches = {} }
|
||||||
for _, marker in ipairs(markers or {}) do
|
for _, marker in ipairs(markers or {}) do
|
||||||
local project = MARKER_PROJECTORS[marker.kind]
|
local project = MARKER_PROJECTORS[marker.kind]
|
||||||
if project then project(state, marker) end
|
if project then project(state, marker) end
|
||||||
end
|
end
|
||||||
return state.labels, state.branches
|
return state.labels, state.branches
|
||||||
end
|
end
|
||||||
@@ -109,7 +115,6 @@ end
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- Compute branch offsets per consuming instruction.
|
--- Compute branch offsets per consuming instruction.
|
||||||
---
|
|
||||||
--- Disposition table:
|
--- Disposition table:
|
||||||
--- `branch_*` -> relative offset: `target_word - branch_word - 1` (MIPS branch-immediate encoding).
|
--- `branch_*` -> relative offset: `target_word - branch_word - 1` (MIPS branch-immediate encoding).
|
||||||
--- `jump` / `call_addr` -> same value as `branch_*` (a relative word offset).
|
--- `jump` / `call_addr` -> same value as `branch_*` (a relative word offset).
|
||||||
@@ -121,7 +126,7 @@ end
|
|||||||
---
|
---
|
||||||
--- Top-level `atom_offset(F, T)` markers (where the marker is the entire token — `consuming_encoder` == nil) default to `branch_*` behavior (relative offset).
|
--- Top-level `atom_offset(F, T)` markers (where the marker is the entire token — `consuming_encoder` == nil) default to `branch_*` behavior (relative offset).
|
||||||
--- This preserves backward compatibility for any top-level marker that may exist outside a control-transfer instruction.
|
--- This preserves backward compatibility for any top-level marker that may exist outside a control-transfer instruction.
|
||||||
--- @param labels table<string, integer>
|
--- @param labels table<string, integer>
|
||||||
--- @param branches table[]
|
--- @param branches table[]
|
||||||
--- @return BranchOffset[]
|
--- @return BranchOffset[]
|
||||||
local function compute_offsets(labels, branches)
|
local function compute_offsets(labels, branches)
|
||||||
@@ -195,44 +200,48 @@ local function emit_atom_offsets(add, atom)
|
|||||||
add("")
|
add("")
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Generate the per-source .offsets.h header.
|
--- Generate the per-directory .offsets.h header.
|
||||||
--- @param source_path string
|
--- @param dir string -- the absolute source directory
|
||||||
--- @param atoms_data AtomData[]
|
--- @param sources table[] -- sources contributing to this directory (for the header comment)
|
||||||
|
--- @param atoms_data AtomData[]
|
||||||
--- @return string
|
--- @return string
|
||||||
local function generate_header(source_path, atoms_data)
|
local function generate_header(dir, sources, atoms_data)
|
||||||
local basename = duffle.basename_no_ext(source_path)
|
local dir_basename = duffle.basename_no_ext(dir)
|
||||||
|
|
||||||
local lines = {}
|
local lines = {}
|
||||||
local function add(s) lines[#lines + 1] = s end
|
local function add(s) lines[#lines + 1] = s end
|
||||||
|
|
||||||
add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT")
|
add("// Auto-generated by ps1_meta.lua (passes/offsets.lua) — DO NOT EDIT")
|
||||||
add("// Source: " .. source_path)
|
add("// Directory: " .. dir:gsub("/", "\\") .. "\\")
|
||||||
|
for _, src in ipairs(sources) do
|
||||||
|
add("// source: " .. src.path:gsub("/", "\\"))
|
||||||
|
end
|
||||||
add("#pragma once")
|
add("#pragma once")
|
||||||
add("")
|
add("")
|
||||||
add("#pragma region " .. basename)
|
add("#pragma region " .. dir_basename)
|
||||||
add("")
|
add("")
|
||||||
add("")
|
add("")
|
||||||
for _, atom in ipairs(atoms_data) do
|
for _, atom in ipairs(atoms_data) do
|
||||||
emit_atom_offsets(add, atom)
|
emit_atom_offsets(add, atom)
|
||||||
end
|
end
|
||||||
add("#pragma endregion " .. basename)
|
add("#pragma endregion " .. dir_basename)
|
||||||
add("")
|
add("")
|
||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
local M = {}
|
local M = {}
|
||||||
|
|
||||||
--- (internal) Process one source: render offsets from canonical atom paths.
|
--- (internal) Aggregate atoms from every source in one directory, render the per-directory `offsets.h`.
|
||||||
--- Returns the offsets_h path if a header was written, or nil.
|
--- Returns the offsets_h path if a header was written, or nil.
|
||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @param src SourceFile
|
--- @param dir string -- the absolute source directory
|
||||||
|
--- @param sources SourceFile[] -- sources in this directory
|
||||||
--- @return string|nil -- the offsets_h path
|
--- @return string|nil -- the offsets_h path
|
||||||
local function process_source(ctx, src)
|
local function process_directory(ctx, dir, sources)
|
||||||
local atoms_data = {}
|
local atoms_data = {}
|
||||||
local scan = src.scan or {}
|
|
||||||
|
|
||||||
local function append_atom(atom)
|
local function append_atom(atom)
|
||||||
local paths = atom and atom.paths
|
local paths = atom and atom.paths
|
||||||
if not paths then return end
|
if not paths then return end
|
||||||
local labels, branches = project_markers(paths.markers)
|
local labels, branches = project_markers(paths.markers)
|
||||||
atoms_data[#atoms_data + 1] = {
|
atoms_data[#atoms_data + 1] = {
|
||||||
@@ -242,19 +251,22 @@ local function process_source(ctx, src)
|
|||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end
|
for _, src in ipairs(sources) do
|
||||||
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end
|
local scan = src.scan or {}
|
||||||
|
for _, atom in ipairs(scan.atoms or {}) do append_atom(atom) end
|
||||||
|
for _, atom in ipairs(scan.raw_atoms or {}) do append_atom(atom) end
|
||||||
|
end
|
||||||
if #atoms_data == 0 then return nil end
|
if #atoms_data == 0 then return nil end
|
||||||
|
|
||||||
local out_path = src.dir .. "/gen/" .. duffle.basename_no_ext(src.dir) .. ".offsets.h"
|
local out_path = dir .. "/gen/offsets.h"
|
||||||
duffle.ensure_dir(duffle.dirname(out_path))
|
duffle.ensure_dir(duffle.dirname(out_path))
|
||||||
duffle.write_file(out_path, generate_header(src.path:gsub("/", "\\"), atoms_data))
|
duffle.write_file(out_path, generate_header(dir, sources, atoms_data))
|
||||||
return out_path
|
return out_path
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Run the offsets pass.
|
--- Run the offsets pass.
|
||||||
--- For each canonical source, emits a per-module `<dir_basename>.offsets.h`
|
--- For each canonical source-directory, emits a per-directory `gen/offsets.h`
|
||||||
--- containing constants for every marker recorded in atom.paths.
|
--- containing constants for every marker recorded in atom.paths across every source in that directory.
|
||||||
--- @param ctx PassCtx
|
--- @param ctx PassCtx
|
||||||
--- @return PassResult
|
--- @return PassResult
|
||||||
function M.run(ctx)
|
function M.run(ctx)
|
||||||
@@ -263,12 +275,17 @@ function M.run(ctx)
|
|||||||
local warnings = {}
|
local warnings = {}
|
||||||
|
|
||||||
local corpus = ctx.shared and ctx.shared.corpus
|
local corpus = ctx.shared and ctx.shared.corpus
|
||||||
if type(corpus) ~= "table" or type(corpus.source_order) ~= "table" then
|
if type(corpus) ~= "table" then
|
||||||
error("offsets.run requires ctx.shared.corpus.source_order (canonical corpus).", 0)
|
error("offsets.run requires ctx.shared.corpus", 0)
|
||||||
|
end
|
||||||
|
if type(corpus.source_order) ~= "table" then
|
||||||
|
error("offsets.run requires ctx.shared.corpus.source_order.", 0)
|
||||||
end
|
end
|
||||||
|
|
||||||
for _, src in ipairs(corpus.source_order) do
|
-- Per-directory aggregation: every source in the same directory contributes to one `gen/offsets.h`.
|
||||||
local out_path = process_source(ctx, src)
|
local sources_by_dir = corpus.sources_by_dir or duffle.group_sources_by_dir(corpus.source_order)
|
||||||
|
for dir, sources in pairs(sources_by_dir) do
|
||||||
|
local out_path = process_directory(ctx, dir, sources)
|
||||||
if out_path then
|
if out_path then
|
||||||
outputs[#outputs + 1] = { offsets_h = out_path }
|
outputs[#outputs + 1] = { offsets_h = out_path }
|
||||||
end
|
end
|
||||||
|
|||||||
+77
-76
@@ -1,22 +1,21 @@
|
|||||||
--- passes/report.lua — Per-MODULE annotation report renderer +
|
--- passes/report.lua — Per-MODULE annotation report renderer + project-wide summary writer.
|
||||||
--- project-wide summary writer.
|
|
||||||
---
|
---
|
||||||
--- Two output files per build:
|
--- Two output files per build:
|
||||||
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
--- - `build/gen/<dir_basename>.annotations.txt` — one per source-directory containing atoms; aggregates across all sources in the directory.
|
||||||
--- - `build/gen/annotation_validation.txt` — the project summary.
|
--- - `build/gen/annotation_validation.txt` — the project summary.
|
||||||
---
|
---
|
||||||
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
|
--- The annotation pass emits `errors.h` files per module and the canonical `corpus.sources_by_dir` projection groups sources by directory.
|
||||||
--- This pass iterates the canonical dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
|
--- This pass iterates the dir projection directly and re-validates each source via `annotation.validate()` to get the detailed per-source results.
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Module-scope requires + package.path setup
|
-- Module-scope requires + package.path setup
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
-- Resolve `arg[0]` to an absolute-ish script directory so that `require("duffle")` resolves against `scripts/` regardless of CWD.
|
||||||
-- Bootstrap: see `ps1_meta.lua` for the rationale.
|
-- Bootstrap: See `ps1_meta.lua` for the rationale.
|
||||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
-- Bootstrap: Load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||||
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
-- Uses `debug.getinfo` to find this file's own directory, so it works both standalone and when require'd from the orchestrator.
|
||||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
-- Bootstrap: Load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when require'd).
|
||||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
@@ -37,7 +36,7 @@ local atoms_source_map = dofile(_bootstrap_dir .. "atoms_source_map.lua")
|
|||||||
|
|
||||||
-- Section separators used in the rendered text reports.
|
-- Section separators used in the rendered text reports.
|
||||||
-- The thin rules are hand-tuned to align with the per-section content width; do not change without also checking the section renderers below.
|
-- The thin rules are hand-tuned to align with the per-section content width; do not change without also checking the section renderers below.
|
||||||
local RULE_THICK = "========================================================"
|
local RULE_THICK = "========================================================"
|
||||||
local SECTION_HEADER_ATOMS = "── Atoms ────────────────────────────────────────────────"
|
local SECTION_HEADER_ATOMS = "── Atoms ────────────────────────────────────────────────"
|
||||||
local SECTION_HEADER_ANNOTS = "── Annotations ──────────────────────────────────────────"
|
local SECTION_HEADER_ANNOTS = "── Annotations ──────────────────────────────────────────"
|
||||||
local SECTION_HEADER_BINDS = "── Binds_* structs ──────────────────────────────────────"
|
local SECTION_HEADER_BINDS = "── Binds_* structs ──────────────────────────────────────"
|
||||||
@@ -59,83 +58,83 @@ local PASS_NAME = "report"
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field sources SourceFile[] -- all source files in the build
|
--- @field sources SourceFile[] -- All source files in the build
|
||||||
--- @field metadata_path string -- path to word_count.metadata.h
|
--- @field metadata_path string -- Path to word_count.metadata.h
|
||||||
--- @field shared table -- cross-pass shared state
|
--- @field shared table -- Cross-pass shared state
|
||||||
--- @field out_root string -- output root (e.g. "build/gen")
|
--- @field out_root string -- Output root (e.g. "build/gen")
|
||||||
--- @field project_root string -- project root (e.g. "code/")
|
--- @field project_root string -- Project root (e.g. "code/")
|
||||||
--- @field upstream table<string, table> -- per-pass upstream outputs
|
--- @field upstream table<string, table> -- Per-pass upstream outputs
|
||||||
--- @field flags table -- CLI flags + per-pass stash
|
--- @field flags table -- CLI flags + per-pass stash
|
||||||
--- @field verbose boolean -- if true, log diagnostic info
|
--- @field verbose boolean -- If true, log diagnostic info
|
||||||
|
|
||||||
--- @class PassResult
|
--- @class PassResult
|
||||||
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
--- @field outputs table[] -- {kind=, path=} entries describing emit files
|
||||||
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
--- @field errors table[] -- {line=, msg=} entries; build-stops
|
||||||
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
--- @field warnings table[] -- {line=, msg=} entries; build-succeeds
|
||||||
|
|
||||||
-- Shapes produced by `passes/annotation.lua`'s `M.validate()`.
|
-- Shapes produced by `passes/annotation.lua`'s `M.validate()`.
|
||||||
|
|
||||||
--- @class AtomEntry
|
--- @class AtomEntry
|
||||||
--- @field name string -- atom name (e.g. "cube_g4_face")
|
--- @field name string -- Atom name (e.g. "cube_g4_face")
|
||||||
--- @field line integer -- source line of the atom declaration
|
--- @field line integer -- Source line of the atom declaration
|
||||||
|
|
||||||
--- @class AnnotEntry
|
--- @class AnnotEntry
|
||||||
--- @field line integer -- source line
|
--- @field line integer -- Source line
|
||||||
--- @field macro string -- the macro name (e.g. "atom_reads")
|
--- @field macro string -- Macro name (e.g. "atom_reads")
|
||||||
--- @field name string -- the atom name (if a `name(...)` was given)
|
--- @field name string -- Atom name (if a `name(...)` was given)
|
||||||
--- @field kind string -- "atom_info" | "atom_bind" | ...
|
--- @field kind string -- "atom_info" | "atom_bind" | ...
|
||||||
--- @field binds string|nil -- Binds_X name if any
|
--- @field binds string|nil -- Binds_X name if any
|
||||||
--- @field reads string[] -- R_* names (read targets)
|
--- @field reads string[] -- R_* names (read targets)
|
||||||
--- @field writes string[] -- R_* names (write targets)
|
--- @field writes string[] -- R_* names (write targets)
|
||||||
--- @field error string|nil -- error message if annotation was malformed
|
--- @field error string|nil -- Error message if annotation was malformed
|
||||||
|
|
||||||
--- @class BindsField
|
--- @class BindsField
|
||||||
--- @field name string -- field name
|
--- @field name string -- Field name
|
||||||
--- @field offset integer -- byte offset within the Binds_X struct
|
--- @field offset integer -- Byte offset within the Binds_X struct
|
||||||
|
|
||||||
--- @class BindsStruct
|
--- @class BindsStruct
|
||||||
--- @field name string -- struct name (e.g. "Binds_Floor")
|
--- @field name string -- Struct name (e.g. "Binds_Floor")
|
||||||
--- @field line integer -- source line of the typedef
|
--- @field line integer -- Source line of the typedef
|
||||||
--- @field bytes integer -- total byte size
|
--- @field bytes integer -- Total byte size
|
||||||
--- @field fields BindsField[] -- the field list
|
--- @field fields BindsField[] -- The field list
|
||||||
|
|
||||||
--- @class MacroEntry
|
--- @class MacroEntry
|
||||||
--- @field name string -- macro name (e.g. "WORD_COUNT(my_macro, 4)")
|
--- @field name string -- Macro name (e.g. "WORD_COUNT(my_macro, 4)")
|
||||||
--- @field line integer -- source line
|
--- @field line integer -- Source line
|
||||||
--- @field words integer -- declared word count
|
--- @field words integer -- Declared word count
|
||||||
|
|
||||||
--- @class Finding
|
--- @class Finding
|
||||||
--- @field line integer -- source line
|
--- @field line integer -- Source line
|
||||||
--- @field msg string -- finding message
|
--- @field msg string -- Finding message
|
||||||
|
|
||||||
--- @class AnnotationResult
|
--- @class AnnotationResult
|
||||||
--- @field source string -- set by this pass; original source path
|
--- @field source string -- Set by this pass; original source path
|
||||||
--- @field atoms AtomEntry[] -- atom declarations in this source
|
--- @field atoms AtomEntry[] -- Atom declarations in this source
|
||||||
--- @field annots AnnotEntry[] -- annotation entries
|
--- @field annots AnnotEntry[] -- Annotation entries
|
||||||
--- @field macros MacroEntry[] -- macro word-count declarations
|
--- @field macros MacroEntry[] -- Macro word-count declarations
|
||||||
--- @field binds BindsStruct[] -- Binds_* struct declarations
|
--- @field binds BindsStruct[] -- Binds_* struct declarations
|
||||||
--- @field errors Finding[] -- errors from validation
|
--- @field errors Finding[] -- Errors from validation
|
||||||
--- @field warnings Finding[] -- warnings from validation
|
--- @field warnings Finding[] -- Warnings from validation
|
||||||
--- @field info table -- info summary (not rendered here)
|
--- @field info table -- Info summary (not rendered here)
|
||||||
|
|
||||||
--- @class ModuleEntry
|
--- @class ModuleEntry
|
||||||
--- @field dir string -- absolute directory path
|
--- @field dir string -- Absolute directory path
|
||||||
--- @field dir_basename string -- basename (e.g. "duffle", "gte_hello")
|
--- @field dir_basename string -- Basename (e.g. "duffle", "gte_hello")
|
||||||
--- @field atoms_count integer -- pre-counted atoms for filtering
|
--- @field atoms_count integer -- Pre-counted atoms for filtering
|
||||||
|
|
||||||
--- @class ModuleReport
|
--- @class ModuleReport
|
||||||
--- @field dir string -- module directory
|
--- @field dir string -- Module directory
|
||||||
--- @field sources SourceFile[] -- sources in this module
|
--- @field sources SourceFile[] -- Sources in this module
|
||||||
--- @field results AnnotationResult[] -- per-source validate() results
|
--- @field results AnnotationResult[] -- Per-source validate() results
|
||||||
|
|
||||||
--- @class ProjectReport
|
--- @class ProjectReport
|
||||||
--- @field results AnnotationResult[] -- all per-source results
|
--- @field results AnnotationResult[] -- All per-source results
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Per-MODULE annotation report (aggregated across all sources in a dir)
|
-- Per-MODULE annotation report (aggregated across all sources in a dir)
|
||||||
@@ -153,9 +152,16 @@ end
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- Render the thin project-wide summary (`build/atom_meta_report.summary.md`).
|
--- Render the thin project-wide summary (`build/atom_meta_report.summary.md`).
|
||||||
--- @param all_results { module:string, atoms:integer, annots:integer, binds:integer,
|
--- @param all_results {
|
||||||
--- macros:integer, findings:integer, errors:integer,
|
--- module:string,
|
||||||
--- warnings:integer, info:integer }[]
|
--- atoms:integer,
|
||||||
|
--- annots:integer,
|
||||||
|
--- binds:integer,
|
||||||
|
--- macros:integer,
|
||||||
|
--- findings:integer,
|
||||||
|
--- errors:integer,
|
||||||
|
--- warnings:integer,
|
||||||
|
--- info:integer }[]
|
||||||
--- @return string
|
--- @return string
|
||||||
local function render_project_summary(all_results)
|
local function render_project_summary(all_results)
|
||||||
local lines = {
|
local lines = {
|
||||||
@@ -165,13 +171,10 @@ local function render_project_summary(all_results)
|
|||||||
"| module | atoms | annots | binds | macros | findings | errors | warnings | info |",
|
"| module | atoms | annots | binds | macros | findings | errors | warnings | info |",
|
||||||
"|--------|-------|--------|-------|--------|----------|--------|----------|------|",
|
"|--------|-------|--------|-------|--------|----------|--------|----------|------|",
|
||||||
}
|
}
|
||||||
local totals = { atoms = 0, annots = 0, binds = 0, macros = 0,
|
local totals = { atoms = 0, annots = 0, binds = 0, macros = 0, findings = 0, errors = 0, warnings = 0, info = 0 }
|
||||||
findings = 0, errors = 0, warnings = 0, info = 0 }
|
|
||||||
for _, e in ipairs(all_results) do
|
for _, e in ipairs(all_results) do
|
||||||
lines[#lines + 1] = string.format(
|
lines[#lines + 1] = string.format("| %s | %d | %d | %d | %d | %d | %d | %d | %d |"
|
||||||
"| %s | %d | %d | %d | %d | %d | %d | %d | %d |",
|
, e.module, e.atoms, e.annots, e.binds, e.macros, e.findings, e.errors, e.warnings, e.info)
|
||||||
e.module, e.atoms, e.annots, e.binds, e.macros,
|
|
||||||
e.findings, e.errors, e.warnings, e.info)
|
|
||||||
totals.atoms = totals.atoms + e.atoms
|
totals.atoms = totals.atoms + e.atoms
|
||||||
totals.annots = totals.annots + e.annots
|
totals.annots = totals.annots + e.annots
|
||||||
totals.binds = totals.binds + e.binds
|
totals.binds = totals.binds + e.binds
|
||||||
@@ -181,23 +184,21 @@ local function render_project_summary(all_results)
|
|||||||
totals.warnings = totals.warnings + e.warnings
|
totals.warnings = totals.warnings + e.warnings
|
||||||
totals.info = totals.info + e.info
|
totals.info = totals.info + e.info
|
||||||
end
|
end
|
||||||
lines[#lines + 1] = string.format(
|
lines[#lines + 1] = string.format("| **TOTAL** | %d | %d | %d | %d | %d | %d | %d | %d |"
|
||||||
"| **TOTAL** | %d | %d | %d | %d | %d | %d | %d | %d |",
|
, totals.atoms, totals.annots, totals.binds, totals.macros, totals.findings, totals.errors, totals.warnings, totals.info)
|
||||||
totals.atoms, totals.annots, totals.binds, totals.macros,
|
|
||||||
totals.findings, totals.errors, totals.warnings, totals.info)
|
|
||||||
return table.concat(lines, "\n") .. "\n"
|
return table.concat(lines, "\n") .. "\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Render the per-module verbose source-map markdown (`build/<module>.atoms.md`).
|
--- Render the per-module verbose source-map markdown (`build/<module>.atoms.md`).
|
||||||
--- Per-source sub-section, per-atom stanza with sourcemap + provenance rows.
|
--- Per-source sub-section, per-atom stanza with sourcemap + provenance rows.
|
||||||
--- Pulls sourcemap + provenance from `atoms_source_map` (no second source walk).
|
--- Pulls sourcemap + provenance from `atoms_source_map` (no second source walk).
|
||||||
--- @param dir string
|
--- @param dir string
|
||||||
--- @param dir_sources SourceFile[]
|
--- @param dir_sources SourceFile[]
|
||||||
--- @param wc table<string, integer>
|
--- @param wc table<string, integer>
|
||||||
--- @return string
|
--- @return string
|
||||||
local function render_module_atoms_md(dir, dir_sources, wc)
|
local function render_module_atoms_md(dir, dir_sources, wc)
|
||||||
local dir_basename = source_basename(dir)
|
local dir_basename = source_basename(dir)
|
||||||
local lines = {
|
local lines = {
|
||||||
"# " .. dir_basename .. " — atoms (verbose source map)",
|
"# " .. dir_basename .. " — atoms (verbose source map)",
|
||||||
"> Per-word call-site + provenance. Auto-generated.",
|
"> Per-word call-site + provenance. Auto-generated.",
|
||||||
"",
|
"",
|
||||||
@@ -249,14 +250,14 @@ end
|
|||||||
--- Aggregates annotation + static-analysis content across all sources in `dir`.
|
--- Aggregates annotation + static-analysis content across all sources in `dir`.
|
||||||
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
|
--- Annotations come from re-running `annotation.validate()` per source (the existing pattern);
|
||||||
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
|
--- static-analysis comes from `corpus.static_analysis_results[dir_basename]` (populated by `static_analysis.lua` — no second corpus_pipe_ctx build).
|
||||||
--- @param dir string
|
--- @param dir string
|
||||||
--- @param dir_sources SourceFile[]
|
--- @param dir_sources SourceFile[]
|
||||||
--- @param annot_results AnnotationResult[]
|
--- @param annot_results AnnotationResult[]
|
||||||
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
|
--- @param sa_results table -- corpus.static_analysis_results[dir_basename]
|
||||||
--- @return string
|
--- @return string
|
||||||
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
local function render_module_meta_report(dir, dir_sources, annot_results, sa_results)
|
||||||
local dir_basename = source_basename(dir)
|
local dir_basename = source_basename(dir)
|
||||||
local lines = {
|
local lines = {
|
||||||
"# " .. dir_basename .. " — atom meta report",
|
"# " .. dir_basename .. " — atom meta report",
|
||||||
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
"> Auto-generated by ps1_meta.lua (passes/report.lua). Do not edit.",
|
||||||
"",
|
"",
|
||||||
@@ -326,8 +327,8 @@ local function render_module_meta_report(dir, dir_sources, annot_results, sa_res
|
|||||||
local binds = a.binds or "—"
|
local binds = a.binds or "—"
|
||||||
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or "—"
|
local reads = (#a.reads > 0 and table.concat(a.reads, ",")) or "—"
|
||||||
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or "—"
|
local writes = (#a.writes > 0 and table.concat(a.writes, ",")) or "—"
|
||||||
add(string.format("| %s | %d | %s | %s | %s | %s |",
|
add(string.format("| %s | %d | %s | %s | %s | %s |"
|
||||||
src_name, a.line, a.name, binds, reads, writes))
|
, src_name, a.line, a.name, binds, reads, writes))
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -418,7 +419,7 @@ local function render_module_meta_report(dir, dir_sources, annot_results, sa_res
|
|||||||
for _, a in ipairs(sorted) do
|
for _, a in ipairs(sorted) do
|
||||||
local p = a.paths or {}
|
local p = a.paths or {}
|
||||||
local src_name = a.source_path and source_basename(a.source_path) or ""
|
local src_name = a.source_path and source_basename(a.source_path) or ""
|
||||||
local notes = ""
|
local notes = ""
|
||||||
if p.has_loops then notes = notes .. " [loop!]" end
|
if p.has_loops then notes = notes .. " [loop!]" end
|
||||||
if p.unknown_macros and #p.unknown_macros > 0 then
|
if p.unknown_macros and #p.unknown_macros > 0 then
|
||||||
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
|
notes = notes .. " [unknown: " .. table.concat(p.unknown_macros, ", ") .. "]"
|
||||||
@@ -426,7 +427,7 @@ local function render_module_meta_report(dir, dir_sources, annot_results, sa_res
|
|||||||
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
|
add(string.format("| %s | %s | %d | %d | %d | %d | %s |",
|
||||||
a.name, src_name,
|
a.name, src_name,
|
||||||
p.cycles_min or 0, p.cycles_max or 0,
|
p.cycles_min or 0, p.cycles_max or 0,
|
||||||
p.branches or 0, p.paths or 0, notes))
|
p.branches or 0, p.paths or 0, notes))
|
||||||
end
|
end
|
||||||
add("")
|
add("")
|
||||||
|
|
||||||
|
|||||||
+376
-167
@@ -2,8 +2,8 @@
|
|||||||
---
|
---
|
||||||
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once,
|
--- Single source-walk pass that produces the fat `SourceScan` payload consumed by all downstream passes. Walks each corpus source record once,
|
||||||
--- extracting every construct type the metaprograms need:
|
--- extracting every construct type the metaprograms need:
|
||||||
---
|
|
||||||
--- MipsAtom_ (kind = "atom", with optional atom_info inner)
|
--- MipsAtom_ (kind = "atom", with optional atom_info inner)
|
||||||
|
--- MipsAtom_Proc_ (kind = "atom_proc", body inside last {})
|
||||||
--- MipsAtomComp_ (kind = "comp_bare")
|
--- MipsAtomComp_ (kind = "comp_bare")
|
||||||
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
|
--- MipsAtomComp_Proc_ (kind = "comp_proc", body inside last {})
|
||||||
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
|
--- atom_dbg_skip — bare whole-atom/component debug-step marker; following declaration disambiguates
|
||||||
@@ -14,8 +14,6 @@
|
|||||||
--- The result is attached to each `src.scan` so downstream passes can read from `src.scan.atoms` / `src.scan.binds` / etc. without re-walking the source.
|
--- The result is attached to each `src.scan` so downstream passes can read from `src.scan.atoms` / `src.scan.binds` / etc. without re-walking the source.
|
||||||
--- This is the first pass in the dep graph (no deps).
|
--- This is the first pass in the dep graph (no deps).
|
||||||
--- Every other pass that reads source structure depends on this one — see `ps1_meta.lua :: PASSES`.
|
--- Every other pass that reads source structure depends on this one — see `ps1_meta.lua :: PASSES`.
|
||||||
---
|
|
||||||
--- **Conventions**: tabs (1/level), EmmyLua annotations, no regex, Lua 5.3 compatible
|
|
||||||
|
|
||||||
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
-- Bootstrap: same as entry scripts. See `ps1_meta.lua` for the rationale.
|
||||||
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
-- Bootstrap: load `scripts/duffle_paths.lua` (sets package.path + package.cpath).
|
||||||
@@ -23,7 +21,7 @@
|
|||||||
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when required).
|
-- Bootstrap: load `duffle_paths.lua` via `debug.getinfo(1, "S").source` (works both standalone + when required).
|
||||||
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
-- duffle_paths.lua sets package.path then returns `require("duffle")` at the bottom, so the dofile value IS the duffle module.
|
||||||
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
local _bootstrap_dir = debug.getinfo(1, "S").source:match("^@?(.*[/\\])") or "./"
|
||||||
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
local duffle = dofile(_bootstrap_dir .. "../duffle_paths.lua")
|
||||||
|
|
||||||
-- Forward declarations for helpers used by earlier parsers (parse_enum_body_fields needs parse_enum_int_literal;
|
-- Forward declarations for helpers used by earlier parsers (parse_enum_body_fields needs parse_enum_int_literal;
|
||||||
-- parse_typedef_binds needs duffle.find_byte).
|
-- parse_typedef_binds needs duffle.find_byte).
|
||||||
@@ -37,7 +35,7 @@ local parse_enum_int_literal
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class SourceScan
|
--- @class SourceScan
|
||||||
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtomComp_ + MipsAtomComp_Proc_
|
--- @field atoms AtomEntry[] -- MipsAtom_ + MipsAtom_Proc_ + MipsAtomComp_ + MipsAtomComp_Proc_
|
||||||
--- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only)
|
--- @field raw_atoms AtomEntry[] -- MipsCode code_<name> { body } (offsets pass only)
|
||||||
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
|
--- @field binds BindsEntry[] -- typedef Struct_(Binds_X) { fields } (fields pre-parsed)
|
||||||
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
|
--- @field atom_infos AtomInfoEntry[] -- MipsAtom_(name) atom_info(...) (sub-calls pre-parsed)
|
||||||
@@ -50,15 +48,15 @@ local parse_enum_int_literal
|
|||||||
--- @field line_of fun(pos: integer): integer -- shared LineIndex closure
|
--- @field line_of fun(pos: integer): integer -- shared LineIndex closure
|
||||||
|
|
||||||
--- @class DebugSkipMarker
|
--- @class DebugSkipMarker
|
||||||
--- @field marker_kind string -- exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive; any other ident reaches the unrelated fallback and is never associated with a declaration.
|
--- @field marker_kind string -- Exact marker ident read from source. Only "atom_dbg_skip" (bare) is positive; any other ident reaches the unrelated fallback and is never associated with a declaration.
|
||||||
--- @field marker_line integer -- line of the marker ident start
|
--- @field marker_line integer -- Line of the marker ident start
|
||||||
--- @field marker_pos integer -- byte position of the marker ident start (the comment walker anchors here)
|
--- @field marker_pos integer -- Byte position of the marker ident start (the comment walker anchors here)
|
||||||
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
|
--- @field is_bare boolean -- true iff marker_kind == "atom_dbg_skip" AND has_parens == false (the only positive form)
|
||||||
--- @field has_parens boolean -- true iff a `(...)` follows the marker ident (diagnostic-only)
|
--- @field has_parens boolean -- true iff a `(...)` follows the marker ident (diagnostic-only)
|
||||||
--- @field args string|nil -- trimmed args inside the `(...)` (nil when has_parens is false)
|
--- @field args string|nil -- Trimmed args inside the `(...)` (nil when has_parens is false)
|
||||||
--- @field pending boolean -- true while awaiting the following declaration
|
--- @field pending boolean -- true while awaiting the following declaration
|
||||||
--- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot
|
--- @field superseded_by_marker_line integer|nil -- set when a newer marker bumped this one out of the pending slot
|
||||||
--- @field target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
|
--- @field target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed (nil if no declaration ever followed)
|
||||||
--- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_`
|
--- @field proc_prelude boolean|nil -- true after the marker crossed an `FI_` prelude and awaits `MipsAtomComp_Proc_`
|
||||||
|
|
||||||
--- @class RegTypeDefault
|
--- @class RegTypeDefault
|
||||||
@@ -71,15 +69,15 @@ local parse_enum_int_literal
|
|||||||
--- @field reg string -- "R_T0"
|
--- @field reg string -- "R_T0"
|
||||||
--- @field type_name string
|
--- @field type_name string
|
||||||
--- @field pointer_depth integer
|
--- @field pointer_depth integer
|
||||||
--- @field source_line integer -- line of the call site (callsite or enum-site)
|
--- @field source_line integer -- Line of the call site (callsite or enum-site)
|
||||||
|
|
||||||
--- @class AtomCtxEntry
|
--- @class AtomCtxEntry
|
||||||
--- @field rbind_atom string -- the rbind atom ident that this consumer should propagate types from
|
--- @field rbind_atom string -- The rbind atom ident that this consumer should propagate types from
|
||||||
--- @field info_line integer
|
--- @field info_line integer
|
||||||
--- @field source string -- absolute path of the source file
|
--- @field source string -- Absolute path of the source file
|
||||||
|
|
||||||
--- @class AtomPhaseGroup
|
--- @class AtomPhaseGroup
|
||||||
--- @field atoms string[] -- atom names tagged with this phase label (source-order)
|
--- @field atoms string[] -- Atom names tagged with this phase label (source-order)
|
||||||
|
|
||||||
--- @class AtomViewEntry
|
--- @class AtomViewEntry
|
||||||
--- @field atom_name string -- e.g. "red_cube_g4_face"
|
--- @field atom_name string -- e.g. "red_cube_g4_face"
|
||||||
@@ -88,11 +86,11 @@ local parse_enum_int_literal
|
|||||||
--- @field info_line integer -- line of the atom_info call
|
--- @field info_line integer -- line of the atom_info call
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
--- @field scan table -- pre-scanned SourceScan payload (set by this pass)
|
--- @field scan table -- Pre-scanned SourceScan payload (set by this pass)
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field sources SourceFile[]
|
--- @field sources SourceFile[]
|
||||||
@@ -111,15 +109,15 @@ local parse_enum_int_literal
|
|||||||
|
|
||||||
--- @class AtomEntry
|
--- @class AtomEntry
|
||||||
--- @field line integer
|
--- @field line integer
|
||||||
--- @field name string -- atom name (for components: without ac_ prefix)
|
--- @field name string -- Atom name (for components: without ac_ prefix)
|
||||||
--- @field body string -- brace-delimited body (without the braces)
|
--- @field body string -- Brace-delimited body (without the braces)
|
||||||
--- @field body_off integer -- char offset of body[1] in source
|
--- @field body_off integer -- Char offset of body[1] in source
|
||||||
--- @field kind string -- "atom" | "comp_bare" | "comp_proc" | "raw_atom"
|
--- @field kind string -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "raw_atom"
|
||||||
--- @field raw_name string -- un-stripped name (for components: with ac_ prefix)
|
--- @field raw_name string -- Un-stripped name (for components: with ac_ prefix)
|
||||||
--- @field ident_pos integer -- position of the MipsAtom_/MipsAtomComp_ ident start
|
--- @field ident_pos integer -- Position of the MipsAtom_/MipsAtomComp_ ident start
|
||||||
--- @field after_paren integer -- position past the closing paren
|
--- @field after_paren integer -- Position past the closing paren
|
||||||
--- @field debug_skip boolean -- true when an `atom_dbg_skip` bare marker immediately precedes this declaration (sole-owner stamp; see push_debug_skip_marker)
|
--- @field debug_skip boolean -- true when an `atom_dbg_skip` bare marker immediately precedes this declaration (sole-owner stamp; see push_debug_skip_marker)
|
||||||
--- @field declaration_comment string|nil -- populated by the scanner (backward walk past the marker, captures contiguous `/* */` or `//` block)
|
--- @field declaration_comment string|nil -- Populated by the scanner (backward walk past the marker, captures contiguous `/* */` or `//` block)
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Local helpers (shared by per-form parsers)
|
-- Local helpers (shared by per-form parsers)
|
||||||
@@ -139,10 +137,20 @@ local QUALIFIER_KEYWORDS = {
|
|||||||
local AC_PREFIX = "ac_"
|
local AC_PREFIX = "ac_"
|
||||||
local AC_PREFIX_LEN = 3
|
local AC_PREFIX_LEN = 3
|
||||||
|
|
||||||
-- Strip the "ac_" prefix from a component name.
|
-- The function-decl keyword that precedes a MipsAtomComp_Proc_ call.
|
||||||
-- Returns the input unchanged if it doesn't start with the prefix.
|
-- Used by the backward walk in duffle.find_function_decl_for.
|
||||||
-- @param raw_name string
|
local SLICE_MIPS_CODE = "Slice_MipsCode"
|
||||||
-- @return string
|
local SLICE_MIPS_CODE_LEN = #SLICE_MIPS_CODE
|
||||||
|
|
||||||
|
-- The return type that precedes a MipsAtom_Proc_ function declaration.
|
||||||
|
-- Used by the backward walk in duffle.find_atom_proc_decl_for.
|
||||||
|
local MIPS_ATOM_PTR = "MipsAtom*"
|
||||||
|
local MIPS_ATOM_PTR_LEN = #MIPS_ATOM_PTR
|
||||||
|
|
||||||
|
--- Strip the "ac_" prefix from a component name.
|
||||||
|
--- Returns the input unchanged if it doesn't start with the prefix.
|
||||||
|
--- @param raw_name string
|
||||||
|
--- @return string
|
||||||
local function strip_ac_prefix(raw_name)
|
local function strip_ac_prefix(raw_name)
|
||||||
if #raw_name > AC_PREFIX_LEN and raw_name:sub(1, AC_PREFIX_LEN) == AC_PREFIX then
|
if #raw_name > AC_PREFIX_LEN and raw_name:sub(1, AC_PREFIX_LEN) == AC_PREFIX then
|
||||||
return raw_name:sub(AC_PREFIX_LEN + 1)
|
return raw_name:sub(AC_PREFIX_LEN + 1)
|
||||||
@@ -156,7 +164,7 @@ end
|
|||||||
local function push_debug_skip_marker(out, marker)
|
local function push_debug_skip_marker(out, marker)
|
||||||
local markers = out.debug_skip_markers
|
local markers = out.debug_skip_markers
|
||||||
local prior = markers[#markers]
|
local prior = markers[#markers]
|
||||||
if prior and prior.pending then
|
if prior and prior.pending then
|
||||||
prior.pending = false
|
prior.pending = false
|
||||||
prior.superseded_by_marker_line = marker.marker_line
|
prior.superseded_by_marker_line = marker.marker_line
|
||||||
end
|
end
|
||||||
@@ -178,25 +186,25 @@ end
|
|||||||
-- Returns (body, after_brace, body_off) on success, or (nil, fallback_pos) on no brace.
|
-- Returns (body, after_brace, body_off) on success, or (nil, fallback_pos) on no brace.
|
||||||
-- `fallback_pos` defaults to `after_paren + 1` (the common "advance by 1" case).
|
-- `fallback_pos` defaults to `after_paren + 1` (the common "advance by 1" case).
|
||||||
local function find_body_braces(source, after_paren, fallback)
|
local function find_body_braces(source, after_paren, fallback)
|
||||||
local brace = duffle.scan_to_char(source, "{", after_paren)
|
local brace = duffle.scan_to_char(source, "{", after_paren)
|
||||||
if not brace then return nil, fallback or (after_paren + 1) end
|
if not brace then return nil, fallback or (after_paren + 1) end
|
||||||
local body, after_brace = duffle.read_braces(source, brace)
|
local body, after_brace = duffle.read_braces(source, brace)
|
||||||
return body, after_brace, brace + 1
|
return body, after_brace, brace + 1
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Walk backward from `start_pos` capturing contiguous `/* */` block(s) and
|
--- Walk backward from `start_pos` capturing contiguous `/* */` block(s) and
|
||||||
-- `//` line(s) that immediately precede it. The caller (preceding_declaration_comment)
|
--- `//` line(s) that immediately precede it. The caller (preceding_declaration_comment)
|
||||||
-- supplies `start_pos` so the walker does not need to detect marker shape or prelude layout.
|
--- supplies `start_pos` so the walker does not need to detect marker shape or prelude layout.
|
||||||
-- The scanner already knows the marker_pos + decl ident_pos and threads that knowledge forward.
|
--- The scanner already knows the marker_pos + decl ident_pos and threads that knowledge forward.
|
||||||
--
|
---
|
||||||
-- The walker captures:
|
--- The walker captures:
|
||||||
-- - Block comment close `*/` followed by walking back to `/*`.
|
--- - Block comment close `*/` followed by walking back to `/*`.
|
||||||
-- - `//` line comments (the line containing the current non-ws position starts with `//`).
|
--- - `//` line comments (the line containing the current non-ws position starts with `//`).
|
||||||
-- It stops at the first non-ws char that does not begin a comment block or line.
|
--- It stops at the first non-ws char that does not begin a comment block or line.
|
||||||
-- Empty string if no comment is adjacent.
|
--- Empty string if no comment is adjacent.
|
||||||
-- @param source string
|
--- @param source string
|
||||||
-- @param start_pos integer -- exclusive upper bound for the captured block
|
--- @param start_pos integer -- exclusive upper bound for the captured block
|
||||||
-- @return string
|
--- @return string
|
||||||
local function preceding_comment_walk_backward(source, start_pos)
|
local function preceding_comment_walk_backward(source, start_pos)
|
||||||
local pieces = {}
|
local pieces = {}
|
||||||
local scan_pos = start_pos
|
local scan_pos = start_pos
|
||||||
@@ -249,13 +257,13 @@ local function preceding_comment_walk_backward(source, start_pos)
|
|||||||
return table.concat(pieces, "\n")
|
return table.concat(pieces, "\n")
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Resolve the start position for the declaration-comment walk.
|
--- Resolve the start position for the declaration-comment walk.
|
||||||
-- When a debug-skip marker is pending, the walker must start from the position immediately before the marker ident
|
--- When a debug-skip marker is pending, the walker must start from the position immediately before the marker ident
|
||||||
-- (so it walks backward past the marker text and any `FI_ MipsAtom ac_X(args)` proc-prelude layout — neither of which is visible if we start from the declaration ident_pos).
|
--- (so it walks backward past the marker text and any `FI_ MipsAtom ac_X(args)` proc-prelude layout — neither of which is visible if we start from the declaration ident_pos).
|
||||||
-- When no marker is pending, the walker starts from the declaration ident_pos directly.
|
--- When no marker is pending, the walker starts from the declaration ident_pos directly.
|
||||||
-- @param pending_marker DebugSkipMarker|nil
|
--- @param pending_marker DebugSkipMarker|nil
|
||||||
-- @param ident_pos integer -- declaration ident position
|
--- @param ident_pos integer -- declaration ident position
|
||||||
-- @return integer
|
--- @return integer
|
||||||
local function comment_walk_start(pending_marker, ident_pos)
|
local function comment_walk_start(pending_marker, ident_pos)
|
||||||
if pending_marker then
|
if pending_marker then
|
||||||
return pending_marker.marker_pos - 1
|
return pending_marker.marker_pos - 1
|
||||||
@@ -263,16 +271,16 @@ local function comment_walk_start(pending_marker, ident_pos)
|
|||||||
return ident_pos - 1
|
return ident_pos - 1
|
||||||
end
|
end
|
||||||
|
|
||||||
-- Attach the pending marker to the next declaration.
|
--- Attach the pending marker to the next declaration.
|
||||||
-- The declaration form disambiguates whole atoms from components; the resolved `debug_skip` is stamped directly on the declaration record
|
--- The declaration form disambiguates whole atoms from components; the resolved `debug_skip` is stamped directly on the declaration record
|
||||||
-- (sole-owner discipline; see push_debug_skip_marker).
|
--- (sole-owner discipline; see push_debug_skip_marker).
|
||||||
--
|
---
|
||||||
-- A marker is POSITIVE (stamps `debug_skip = true` on the declaration) iff:
|
--- A marker is POSITIVE (stamps `debug_skip = true` on the declaration) iff:
|
||||||
-- marker_kind == "atom_dbg_skip" AND is_bare == true
|
--- marker_kind == "atom_dbg_skip" AND is_bare == true
|
||||||
-- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`.
|
--- Any other spelling or shape (parenthesized form, legacy name) is recorded as a raw marker for annotation validation but never stamps `debug_skip`.
|
||||||
-- @param out SourceScan
|
--- @param out SourceScan
|
||||||
-- @param target_kind string|nil -- "atom" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
--- @param target_kind string|nil -- "atom" | "atom_proc" | "comp_bare" | "comp_proc" | "unrelated" once observed
|
||||||
-- @return boolean|nil -- true iff the marker is the positive bare form
|
--- @return boolean|nil -- true iff the marker is the positive bare form
|
||||||
local function attach_debug_skip_marker(out, target_kind)
|
local function attach_debug_skip_marker(out, target_kind)
|
||||||
local markers = out.debug_skip_markers
|
local markers = out.debug_skip_markers
|
||||||
local marker = markers[#markers]
|
local marker = markers[#markers]
|
||||||
@@ -390,13 +398,13 @@ local function walk_body_fields(body, build_field)
|
|||||||
while body_pos <= body_len do
|
while body_pos <= body_len do
|
||||||
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
|
body_pos = duffle.skip_ws_and_cmt(body, body_pos)
|
||||||
if body_pos > body_len then break end
|
if body_pos > body_len then break end
|
||||||
local first, first_end = duffle.read_ident(body, body_pos)
|
local first, first_end = duffle.read_ident(body, body_pos)
|
||||||
if not first then
|
if not first then
|
||||||
body_pos = body_pos + 1
|
body_pos = body_pos + 1
|
||||||
else
|
else
|
||||||
local after_first = duffle.skip_ws_and_cmt(body, first_end)
|
local after_first = duffle.skip_ws_and_cmt(body, first_end)
|
||||||
local result, new_pos = build_field(first, first_end, after_first)
|
local result, new_pos = build_field(first, first_end, after_first)
|
||||||
if result then fields[#fields + 1] = result end
|
if result then fields[#fields + 1] = result end
|
||||||
body_pos = new_pos or first_end
|
body_pos = new_pos or first_end
|
||||||
-- Skip a single trailing `,` or `;`.
|
-- Skip a single trailing `,` or `;`.
|
||||||
if body_pos <= body_len and (body:sub(body_pos, body_pos) == "," or body:sub(body_pos, body_pos) == ";") then
|
if body_pos <= body_len and (body:sub(body_pos, body_pos) == "," or body:sub(body_pos, body_pos) == ";") then
|
||||||
@@ -442,7 +450,7 @@ local function parse_enum_body_fields(body)
|
|||||||
local value
|
local value
|
||||||
local new_pos
|
local new_pos
|
||||||
if body:sub(after_name, after_name) == "=" then
|
if body:sub(after_name, after_name) == "=" then
|
||||||
local val_pos = duffle.skip_ws_and_cmt(body, after_name + 1)
|
local val_pos = duffle.skip_ws_and_cmt(body, after_name + 1)
|
||||||
local v, end_pos = parse_enum_int_literal(body, val_pos)
|
local v, end_pos = parse_enum_int_literal(body, val_pos)
|
||||||
if v ~= nil then
|
if v ~= nil then
|
||||||
value = v
|
value = v
|
||||||
@@ -461,16 +469,16 @@ end
|
|||||||
-- Returns a positive integer byte_size when the chain bottoms out at a builtin, or nil if the chain is broken, exceeds TYPE_CHAIN_MAX_DEPTH, or contains a cycle.
|
-- Returns a positive integer byte_size when the chain bottoms out at a builtin, or nil if the chain is broken, exceeds TYPE_CHAIN_MAX_DEPTH, or contains a cycle.
|
||||||
local function resolve_typedef_byte_size(type_name, type_name_registry, visited, depth)
|
local function resolve_typedef_byte_size(type_name, type_name_registry, visited, depth)
|
||||||
if depth > TYPE_CHAIN_MAX_DEPTH then return nil end
|
if depth > TYPE_CHAIN_MAX_DEPTH then return nil end
|
||||||
if visited[type_name] then return nil end
|
if visited[type_name] then return nil end
|
||||||
visited[type_name] = true
|
visited[type_name] = true
|
||||||
|
|
||||||
-- Check the builtin primitive map FIRST.
|
-- Check the builtin primitive map FIRST.
|
||||||
-- This handles undeclared builtin idents (e.g. `__UINT32_TYPE__` appears as underlying_type in `typedef __UINT32_TYPE__ TSet_(V4_S2);`
|
-- This handles undeclared builtin idents (e.g. `__UINT32_TYPE__` appears as underlying_type in `typedef __UINT32_TYPE__ TSet_(V4_S2);`
|
||||||
-- even though the fixture never declares `__UINT32_TYPE__` itself).
|
-- even though the fixture never declares `__UINT32_TYPE__` itself).
|
||||||
local builtin = BUILTIN_BYTE_SIZES[type_name]
|
local builtin = BUILTIN_BYTE_SIZES[type_name]
|
||||||
if builtin ~= nil then return builtin end
|
if builtin ~= nil then return builtin end
|
||||||
|
|
||||||
local entry = type_name_registry[type_name]
|
local entry = type_name_registry[type_name]
|
||||||
if not entry then return nil end
|
if not entry then return nil end
|
||||||
|
|
||||||
-- Confident: this entry was already resolved by the propagation pass (e.g., a builtin or a struct whose fields are all resolved).
|
-- Confident: this entry was already resolved by the propagation pass (e.g., a builtin or a struct whose fields are all resolved).
|
||||||
@@ -519,9 +527,9 @@ local function propagate_type_sizes(out)
|
|||||||
for name, entry in pairs(reg) do
|
for name, entry in pairs(reg) do
|
||||||
if entry.byte_size == nil then
|
if entry.byte_size == nil then
|
||||||
local resolved = resolve_typedef_byte_size(name, reg, {}, 1)
|
local resolved = resolve_typedef_byte_size(name, reg, {}, 1)
|
||||||
if resolved ~= nil then
|
if resolved ~= nil then
|
||||||
entry.byte_size = resolved
|
entry.byte_size = resolved
|
||||||
any_change = true
|
any_change = true
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -802,6 +810,11 @@ local BYTE_x = 0x78 -- 'x'
|
|||||||
local BYTE_X = 0x58 -- 'X'
|
local BYTE_X = 0x58 -- 'X'
|
||||||
local BYTE_OPEN_BRACE = 0x7B -- '{'
|
local BYTE_OPEN_BRACE = 0x7B -- '{'
|
||||||
local BYTE_CLOSE_BRACE= 0x7D -- '}'
|
local BYTE_CLOSE_BRACE= 0x7D -- '}'
|
||||||
|
local BYTE_SLASH = 0x2F -- '/'
|
||||||
|
local BYTE_STAR = 0x2A -- '*'
|
||||||
|
local BYTE_SPACE = 0x20 -- ' '
|
||||||
|
local BYTE_TAB = 0x09 -- '\t'
|
||||||
|
local BYTE_CR = 0x0D -- '\r'
|
||||||
|
|
||||||
-- Maximum chain depth when resolving `R_*_Code` symbol RHS references.
|
-- Maximum chain depth when resolving `R_*_Code` symbol RHS references.
|
||||||
-- Eight hops is enough for any production chain (R_TapePtr_Code -> R_T8_Code -> ...).
|
-- Eight hops is enough for any production chain (R_TapePtr_Code -> R_T8_Code -> ...).
|
||||||
@@ -825,10 +838,48 @@ local function hex_digit_value(b)
|
|||||||
return nil
|
return nil
|
||||||
end
|
end
|
||||||
|
|
||||||
|
-- Read one trailing C-comment that appears immediately after `pos` in `body`,
|
||||||
|
-- skipping horizontal whitespace and newlines first. Used by `parse_enum_entry` to
|
||||||
|
-- recover the `atom_auto_reg:` / `phase_auto_reg:` scope annotation embedded by
|
||||||
|
-- the `atom_auto_reg` / `phase_auto_reg` macros' RHS expansion
|
||||||
|
-- (`R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`).
|
||||||
|
-- Handles both block (`/* ... */`) and line (`// ...`) forms.
|
||||||
|
-- Returns the comment text (without delimiters), or nil if no comment is adjacent.
|
||||||
|
local function read_trailing_cmt_after(body, pos)
|
||||||
|
local body_len = #body
|
||||||
|
while pos <= body_len do
|
||||||
|
local b = body:byte(pos)
|
||||||
|
if b == BYTE_SPACE or b == BYTE_TAB or b == BYTE_NEWLINE or b == BYTE_CR then
|
||||||
|
pos = pos + 1
|
||||||
|
elseif b == BYTE_SLASH then
|
||||||
|
local b2 = body:byte(pos + 1)
|
||||||
|
if b2 == BYTE_STAR then
|
||||||
|
-- Block comment /* ... */
|
||||||
|
local i = pos + 2
|
||||||
|
while i < body_len do
|
||||||
|
if body:byte(i) == BYTE_STAR and body:byte(i + 1) == BYTE_SLASH then
|
||||||
|
return body:sub(pos + 2, i - 1)
|
||||||
|
end
|
||||||
|
i = i + 1
|
||||||
|
end
|
||||||
|
return nil -- unterminated; treat as no comment
|
||||||
|
elseif b2 == BYTE_SLASH then
|
||||||
|
-- Line comment // ... (strip the trailing newline)
|
||||||
|
local end_pos = duffle.find_byte(body, BYTE_NEWLINE, pos + 2) or (body_len + 1)
|
||||||
|
return body:sub(pos + 2, end_pos - 1)
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
else
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
--- Parse a decimal/negative-decimal/hex integer literal starting at byte position `start`.
|
--- Parse a decimal/negative-decimal/hex integer literal starting at byte position `start`.
|
||||||
--- Returns (value, end_pos) on success, or (nil, start) on failure / no match.
|
--- Returns (value, end_pos) on success, or (nil, start) on failure / no match.
|
||||||
--- Accepts: 12, -1, 0, 0x10, 0X1F, -0x10.
|
--- Accepts: 12, -1, 0, 0x10, 0X1F, -0x10.
|
||||||
--- @param text string
|
--- @param text string
|
||||||
--- @param start integer
|
--- @param start integer
|
||||||
--- @return integer|nil, integer
|
--- @return integer|nil, integer
|
||||||
--- Implementation note: this is a plain assignment (not `local function`)
|
--- Implementation note: this is a plain assignment (not `local function`)
|
||||||
@@ -952,9 +1003,9 @@ end
|
|||||||
--- Always saves the raw RHS text into `code_macro_bodies` (for cross-source fallback during chain resolution),
|
--- Always saves the raw RHS text into `code_macro_bodies` (for cross-source fallback during chain resolution),
|
||||||
--- then (if resolvable) stores the resolved integer code into `code_macros` keyed by the macro name.
|
--- then (if resolvable) stores the resolved integer code into `code_macros` keyed by the macro name.
|
||||||
--- `directive_start` points at the `#` byte. The function is silent on non-matching directives, the caller skips the line in any case.
|
--- `directive_start` points at the `#` byte. The function is silent on non-matching directives, the caller skips the line in any case.
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param directive_start integer -- byte position of `#`
|
--- @param directive_start integer -- byte position of `#`
|
||||||
--- @param code_macros table -- out._code_macros / ctx.shared._code_macros
|
--- @param code_macros table -- out._code_macros / ctx.shared._code_macros
|
||||||
--- @param code_macro_bodies table -- out._code_macro_bodies / ctx.shared._code_macro_bodies
|
--- @param code_macro_bodies table -- out._code_macro_bodies / ctx.shared._code_macro_bodies
|
||||||
local function try_extract_code_macro(source, directive_start, code_macros, code_macro_bodies)
|
local function try_extract_code_macro(source, directive_start, code_macros, code_macro_bodies)
|
||||||
local rest = duffle.skip_ws_and_cmt(source, directive_start + 1)
|
local rest = duffle.skip_ws_and_cmt(source, directive_start + 1)
|
||||||
@@ -985,8 +1036,8 @@ end
|
|||||||
--- Populates `code_macros` with resolved integer codes AND `code_macro_bodies` with raw RHS text
|
--- Populates `code_macros` with resolved integer codes AND `code_macro_bodies` with raw RHS text
|
||||||
--- (used by the chain walker as cross-source fallback during pass 1b in `M.run`); ignores everything else.
|
--- (used by the chain walker as cross-source fallback during pass 1b in `M.run`); ignores everything else.
|
||||||
--- Used by `M.run` pass 1a to build the cross-source `_code_macros` + `_code_macro_bodies` registries before pass 1b resolves chains.
|
--- Used by `M.run` pass 1a to build the cross-source `_code_macros` + `_code_macro_bodies` registries before pass 1b resolves chains.
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param code_macros table
|
--- @param code_macros table
|
||||||
--- @param code_macro_bodies table
|
--- @param code_macro_bodies table
|
||||||
local function scan_source_pre_pass(source, code_macros, code_macro_bodies)
|
local function scan_source_pre_pass(source, code_macros, code_macro_bodies)
|
||||||
local pos = 1
|
local pos = 1
|
||||||
@@ -1043,7 +1094,7 @@ local function parse_enum_atom_type_default(body, pos)
|
|||||||
if pos > #body then return nil, 0, pos end
|
if pos > #body then return nil, 0, pos end
|
||||||
-- Bare `atom_type` word with word-bounding on both sides.
|
-- Bare `atom_type` word with word-bounding on both sides.
|
||||||
local ident, ident_end = duffle.read_ident(body, pos)
|
local ident, ident_end = duffle.read_ident(body, pos)
|
||||||
if ident ~= "atom_type" then return nil, 0, pos end
|
if ident ~= "atom_type" then return nil, 0, pos end
|
||||||
if pos > 1 then
|
if pos > 1 then
|
||||||
local prev = body:byte(pos - 1)
|
local prev = body:byte(pos - 1)
|
||||||
if duffle.is_alnum_byte(prev) then return nil, 0, pos end
|
if duffle.is_alnum_byte(prev) then return nil, 0, pos end
|
||||||
@@ -1054,19 +1105,19 @@ local function parse_enum_atom_type_default(body, pos)
|
|||||||
end
|
end
|
||||||
-- Expect `( ... )` immediately after.
|
-- Expect `( ... )` immediately after.
|
||||||
local open_pos = duffle.skip_ws_and_cmt(body, ident_end)
|
local open_pos = duffle.skip_ws_and_cmt(body, ident_end)
|
||||||
if open_pos > #body or body:sub(open_pos, open_pos) ~= "(" then return nil, 0, pos end
|
if open_pos > #body or body:sub(open_pos, open_pos) ~= "(" then return nil, 0, pos end
|
||||||
local inner, after_close = duffle.read_parens(body, open_pos)
|
local inner, after_close = duffle.read_parens(body, open_pos)
|
||||||
-- Reject any trailing tokens past the close paren other than comma / close-brace (next enum entry / end of enum).
|
-- Reject any trailing tokens past the close paren other than comma / close-brace (next enum entry / end of enum).
|
||||||
local residue = duffle.skip_ws_and_cmt(body, after_close)
|
local residue = duffle.skip_ws_and_cmt(body, after_close)
|
||||||
if residue <= #body then
|
if residue <= #body then
|
||||||
local rbyte = body:byte(residue)
|
local rbyte = body:byte(residue)
|
||||||
if rbyte ~= BYTE_COMMA and rbyte ~= BYTE_CLOSE_BRACE then return nil, 0, pos end
|
if rbyte ~= BYTE_COMMA and rbyte ~= BYTE_CLOSE_BRACE then return nil, 0, pos end
|
||||||
end
|
end
|
||||||
-- Parse the type chain inside the parens (e.g. `V4_S2*` -> ("V4_S2", 1)).
|
-- Parse the type chain inside the parens (e.g. `V4_S2*` -> ("V4_S2", 1)).
|
||||||
local type_name, depth, after_chain = parse_type_chain(inner, 1)
|
local type_name, depth, after_chain = parse_type_chain(inner, 1)
|
||||||
if not type_name then return nil, 0, pos end
|
if not type_name then return nil, 0, pos end
|
||||||
local end_check = duffle.skip_ws_and_cmt(inner, after_chain)
|
local end_check = duffle.skip_ws_and_cmt(inner, after_chain)
|
||||||
if end_check <= #inner then return nil, 0, pos end
|
if end_check <= #inner then return nil, 0, pos end
|
||||||
return type_name, depth, duffle.skip_ws_and_cmt(body, after_close)
|
return type_name, depth, duffle.skip_ws_and_cmt(body, after_close)
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -1098,11 +1149,11 @@ end
|
|||||||
---
|
---
|
||||||
--- Diagnostic-only path: a following `(...)` is recorded as an invalid parenthesized-form marker so the annotation rule can emit a precise "parenthesized form" diagnostic.
|
--- Diagnostic-only path: a following `(...)` is recorded as an invalid parenthesized-form marker so the annotation rule can emit a precise "parenthesized form" diagnostic.
|
||||||
--- The parenthesized form stays diagnostic; the bare form alone carries the runtime stamp.
|
--- The parenthesized form stays diagnostic; the bare form alone carries the runtime stamp.
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param ident_end integer
|
--- @param ident_end integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
--- @return integer -- source cursor position to resume from
|
--- @return integer -- source cursor position to resume from
|
||||||
local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
|
local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
|
||||||
local marker_kind = source:sub(pos, ident_end - 1)
|
local marker_kind = source:sub(pos, ident_end - 1)
|
||||||
@@ -1131,18 +1182,58 @@ local function parse_dbg_skip_marker(source, pos, ident_end, line_of, out)
|
|||||||
return marker_end
|
return marker_end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
--- Parse `atom_auto_reg(<atom>, R_<Sym>)` and `phase_auto_reg(<phase>, R_<Sym>)` markers.
|
||||||
|
---
|
||||||
|
--- The macros expand to `sym = sym##_Code` per their definition in dsl.atom.h.
|
||||||
|
--- After preprocessing, the marker renders as a full enum entry of the form `R_<Sym> = R_<Sym>_Code,`.
|
||||||
|
--- This parser detects the macro invocation site, extracts `(scope_name, sym)`, and stores it
|
||||||
|
--- in the per-source table (atom_auto_regs or phase_auto_regs) under the scope's name.
|
||||||
|
---
|
||||||
|
--- @param source string
|
||||||
|
--- @param pos integer
|
||||||
|
--- @param ident_end integer
|
||||||
|
--- @param line_of fun(pos: integer): integer
|
||||||
|
--- @param out SourceScan
|
||||||
|
--- @return integer
|
||||||
|
local function parse_auto_reg_marker(source, pos, ident_end, line_of, out)
|
||||||
|
local marker_kind = source:sub(pos, ident_end - 1) -- "atom_auto_reg" or "phase_auto_reg"
|
||||||
|
local scope_kind = marker_kind == "atom_auto_reg" and "atom" or "phase"
|
||||||
|
|
||||||
|
local inner, after_paren = read_parens_after(source, ident_end)
|
||||||
|
if not inner then return after_paren end
|
||||||
|
|
||||||
|
local args = duffle.split_top_level_commas(inner)
|
||||||
|
local scope_name = args[1] and duffle.trim(args[1]) or nil
|
||||||
|
local sym = args[2] and duffle.trim(args[2]) or nil
|
||||||
|
|
||||||
|
-- Filter: only accept `R_<Sym>` form (matches `^R_[%w_]+$`).
|
||||||
|
if scope_name and sym and sym:match("^R_[%w_]+$") then
|
||||||
|
if scope_kind == "atom" then
|
||||||
|
out.atom_auto_regs = out.atom_auto_regs or {}
|
||||||
|
out.atom_auto_regs[scope_name] = out.atom_auto_regs[scope_name] or {}
|
||||||
|
out.atom_auto_regs[scope_name][sym] = sym
|
||||||
|
else
|
||||||
|
out.phase_auto_regs = out.phase_auto_regs or {}
|
||||||
|
out.phase_auto_regs[scope_name] = out.phase_auto_regs[scope_name] or {}
|
||||||
|
out.phase_auto_regs[scope_name][sym] = sym
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return after_paren
|
||||||
|
end
|
||||||
|
|
||||||
-- Parse `atom_dbg_reg_default(R_X, <type>...)`;
|
-- Parse `atom_dbg_reg_default(R_X, <type>...)`;
|
||||||
-- the second argument may be a `Type` or `Type*`/`Type**` chain. Records in `out.types[R_X]`.
|
-- the second argument may be a `Type` or `Type*`/`Type**` chain. Records in `out.types[R_X]`.
|
||||||
local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
|
local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
|
||||||
local inner, after_paren = read_parens_after(source, ident_end)
|
local inner, after_paren = read_parens_after(source, ident_end)
|
||||||
if not inner then return after_paren end
|
if not inner then return after_paren end
|
||||||
local args = duffle.split_top_level_commas(inner)
|
local args = duffle.split_top_level_commas(inner)
|
||||||
if #args < 1 then
|
if #args < 1 then
|
||||||
-- Annotation pass surfaces this; we still consume the marker.
|
-- Annotation pass surfaces this; we still consume the marker.
|
||||||
return after_paren
|
return after_paren
|
||||||
end
|
end
|
||||||
local reg_name = duffle.trim(args[1])
|
local reg_name = duffle.trim(args[1])
|
||||||
local type_part = args[2] or "void"
|
local type_part = args[2] or "void"
|
||||||
local type_name, depth = parse_type_chain(type_part, 1)
|
local type_name, depth = parse_type_chain(type_part, 1)
|
||||||
if not type_name then type_name, depth = duffle.trim(type_part), 0 end
|
if not type_name then type_name, depth = duffle.trim(type_part), 0 end
|
||||||
out.types[reg_name] = {
|
out.types[reg_name] = {
|
||||||
@@ -1161,23 +1252,23 @@ local function parse_atom_dbg_reg_default(source, pos, ident_end, line_of, out)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
|
--- Parse: `MipsAtom_(<name>) [atom_info(<binds>, <reads>, <writes>)] { <body> }`
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param ident_end integer
|
--- @param ident_end integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||||
if not inner then return after_paren end
|
if not inner then return after_paren end
|
||||||
|
|
||||||
local raw_name = duffle.read_ident(inner, 1)
|
local raw_name = duffle.read_ident(inner, 1)
|
||||||
|
|
||||||
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
|
-- Lookahead for atom_info(...) between `)` and `{`. Captures sub-calls; updates brace search start.
|
||||||
local brace_search_pos = after_paren
|
local brace_search_pos = after_paren
|
||||||
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
|
local lookahead = duffle.skip_ws_and_cmt(source, after_paren)
|
||||||
local look_ident, look_end = duffle.read_ident(source, lookahead)
|
local look_ident, look_end = duffle.read_ident(source, lookahead)
|
||||||
if look_ident == "atom_info" then
|
if look_ident == "atom_info" then
|
||||||
local info_open = duffle.skip_ws_and_cmt(source, look_end)
|
local info_open = duffle.skip_ws_and_cmt(source, look_end)
|
||||||
if source:sub(info_open, info_open) == "(" then
|
if source:sub(info_open, info_open) == "(" then
|
||||||
local info_inner, info_after = duffle.read_parens(source, info_open)
|
local info_inner, info_after = duffle.read_parens(source, info_open)
|
||||||
@@ -1221,7 +1312,7 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
|
local body, after_brace, body_off = find_body_braces(source, brace_search_pos, open_paren + 1)
|
||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
if raw_name and raw_name ~= "" then
|
if raw_name and raw_name ~= "" then
|
||||||
register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source)
|
register_atom(out, "atom", line_of(pos), raw_name, body, body_off, raw_name, pos, after_paren, source)
|
||||||
@@ -1231,20 +1322,20 @@ local function parse_mips_atom(source, pos, ident_end, line_of, out)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsAtomComp_(<name>) { <body> }`
|
--- Parse: `MipsAtomComp_(<name>) { <body> }`
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param ident_end integer
|
--- @param ident_end integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
|
local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||||
if not inner then return after_paren end
|
if not inner then return after_paren end
|
||||||
|
|
||||||
local raw_name = duffle.read_ident(inner, 1)
|
local raw_name = duffle.read_ident(inner, 1)
|
||||||
if not raw_name then return open_paren + 1 end
|
if not raw_name then return open_paren + 1 end
|
||||||
|
|
||||||
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
|
local body, after_brace, body_off = find_body_braces(source, after_paren, open_paren + 1)
|
||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
local name = strip_ac_prefix(raw_name)
|
local name = strip_ac_prefix(raw_name)
|
||||||
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
register_atom(out, "comp_bare", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
||||||
@@ -1253,14 +1344,14 @@ local function parse_mips_atom_comp(source, pos, ident_end, line_of, out)
|
|||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsAtomComp_Proc_(<name>, { <body> })` — body is inside the LAST `{` in args.
|
--- Parse: `MipsAtomComp_Proc_(<name>, { <body> })` — body is inside the LAST `{` in args.
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param ident_end integer
|
--- @param ident_end integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
|
local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||||
if not inner then return after_paren end
|
if not inner then return after_paren end
|
||||||
|
|
||||||
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
|
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
|
||||||
@@ -1276,22 +1367,71 @@ local function parse_mips_atom_comp_proc(source, pos, ident_end, line_of, out)
|
|||||||
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
||||||
if close_pos > #inner + 1 then return after_paren end
|
if close_pos > #inner + 1 then return after_paren end
|
||||||
|
|
||||||
local raw_name = inner:match("^%s*([%w_]+)") or "?"
|
-- The component name is derived from the preceding function declaration
|
||||||
local name = strip_ac_prefix(raw_name)
|
-- (`FI_ Slice_MipsCode ac_X(...)`), not from the first macro arg (which
|
||||||
|
-- is now `ab`). The backward walk finds the function decl before open_paren.
|
||||||
|
local raw_name = duffle.find_function_decl_for(source, open_paren, SLICE_MIPS_CODE_LEN)
|
||||||
|
if not raw_name then raw_name = "?" end
|
||||||
|
local name = strip_ac_prefix(raw_name)
|
||||||
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
|
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
|
||||||
local body_off = open_paren + 2 + last_brace_pos
|
local body_off = open_paren + 2 + last_brace_pos
|
||||||
|
|
||||||
register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
register_atom(out, "comp_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
||||||
|
|
||||||
return after_paren
|
return after_paren
|
||||||
end
|
end
|
||||||
|
|
||||||
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
|
--- Parse: `MipsAtom_Proc_(<name>, <abuilder>, { <body> })` — body is inside the LAST `{` in args.
|
||||||
--- @param source string
|
--- Per Task 12.10: full support for the runtime-proc atom form. Registers the atom
|
||||||
--- @param pos integer
|
--- with kind `"atom_proc"` so offsets.lua / components.lua can emit
|
||||||
|
--- * `mac_<name>` aliases in `gen/macs.h` (the components pass)
|
||||||
|
--- * `atom_offset__X__Y` defs in `gen/offsets.h` (the offsets pass)
|
||||||
|
--- The atom name is the FIRST ident of the args (the second arg `ab` is the
|
||||||
|
--- atom-builder, not the name). Unlike `MipsAtomComp_Proc_`, there is no `ac_`
|
||||||
|
--- prefix on the symbol — `MipsAtom_Proc_` is the runtime-proc wrapper, so the
|
||||||
|
--- symbol IS the bare atom name (e.g. `normalize_v3s4`, not `ac_normalize_v3s4`).
|
||||||
|
--- @param source string
|
||||||
|
--- @param pos integer
|
||||||
--- @param ident_end integer
|
--- @param ident_end integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
|
--- @return integer
|
||||||
|
local function parse_mips_atom_proc(source, pos, ident_end, line_of, out)
|
||||||
|
local inner, after_paren, open_paren = read_parens_after(source, ident_end)
|
||||||
|
if not inner then return after_paren end
|
||||||
|
|
||||||
|
-- Find the LAST `{` in inner (the body brace, not any potential embedded braces in expressions).
|
||||||
|
local last_brace_pos = nil
|
||||||
|
for search_pos = #inner, 1, -1 do
|
||||||
|
if inner:sub(search_pos, search_pos) == "{" then last_brace_pos = search_pos; break end
|
||||||
|
end
|
||||||
|
if not last_brace_pos then return after_paren end
|
||||||
|
|
||||||
|
-- Use duffle.read_braces to find the matching close brace.
|
||||||
|
-- Uses `read_balanced` for delimiter-depth tracking.
|
||||||
|
-- If close_pos is past the end of inner, the brace didn't match (malformed input); skip.
|
||||||
|
local body, close_pos = duffle.read_braces(inner, last_brace_pos)
|
||||||
|
if close_pos > #inner + 1 then return after_paren end
|
||||||
|
|
||||||
|
-- The atom name is derived from the preceding function declaration
|
||||||
|
-- (`internal MipsAtom* X_proc(...)`), not from the first macro arg (which
|
||||||
|
-- is now `aa`). The backward walk finds the function decl before open_paren
|
||||||
|
-- and strips the `_proc` suffix.
|
||||||
|
local raw_name = duffle.find_atom_proc_decl_for(source, open_paren, MIPS_ATOM_PTR_LEN)
|
||||||
|
if not raw_name then raw_name = "?" end
|
||||||
|
local name = strip_ac_prefix(raw_name)
|
||||||
|
-- Position of body[1] in source = open_paren + 1 (start of inner) + last_brace_pos + 1 (past '{').
|
||||||
|
local body_off = open_paren + 2 + last_brace_pos
|
||||||
|
register_atom(out, "atom_proc", line_of(pos), name, body, body_off, raw_name, pos, after_paren, source)
|
||||||
|
|
||||||
|
return after_paren
|
||||||
|
end
|
||||||
|
|
||||||
|
--- Parse: `MipsCode code_<name> { <body> }` (raw atom form — offsets pass only).
|
||||||
|
--- @param source string
|
||||||
|
--- @param pos integer
|
||||||
|
--- @param ident_end integer
|
||||||
|
--- @param line_of fun(pos: integer): integer
|
||||||
|
--- @param out SourceScan
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function parse_mips_code(source, pos, ident_end, line_of, out)
|
local function parse_mips_code(source, pos, ident_end, line_of, out)
|
||||||
local next_pos = duffle.skip_ws_and_cmt(source, ident_end)
|
local next_pos = duffle.skip_ws_and_cmt(source, ident_end)
|
||||||
@@ -1300,8 +1440,8 @@ local function parse_mips_code(source, pos, ident_end, line_of, out)
|
|||||||
return ident_end
|
return ident_end
|
||||||
end
|
end
|
||||||
|
|
||||||
local atom_name = next_ident:sub(6)
|
local atom_name = next_ident:sub(6)
|
||||||
local body, after_brace, body_off = find_body_braces(source, next_after, ident_end)
|
local body, after_brace, body_off = find_body_braces(source, next_after, ident_end)
|
||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
register_raw_atom(out, line_of(pos), atom_name, body, body_off, atom_name, pos)
|
register_raw_atom(out, line_of(pos), atom_name, body, body_off, atom_name, pos)
|
||||||
|
|
||||||
@@ -1320,13 +1460,13 @@ end
|
|||||||
--- pointer_depth = 0
|
--- pointer_depth = 0
|
||||||
--- } -- byte_size + per-field offset/byte_size set by the propagation pass.
|
--- } -- byte_size + per-field offset/byte_size set by the propagation pass.
|
||||||
--- Also populates `out.binds[]` IFF `name:sub(1, 6) == "Binds_"`.
|
--- Also populates `out.binds[]` IFF `name:sub(1, 6) == "Binds_"`.
|
||||||
--- @param body string
|
--- @param body string
|
||||||
--- @param name string
|
--- @param name string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
local function register_struct_type(body, name, pos, line_of, out)
|
local function register_struct_type(body, name, pos, line_of, out)
|
||||||
local fields = parse_struct_body_fields(body)
|
local fields = parse_struct_body_fields(body)
|
||||||
local source_pos = line_of(pos)
|
local source_pos = line_of(pos)
|
||||||
out.type_name_registry[name] = {
|
out.type_name_registry[name] = {
|
||||||
name = name,
|
name = name,
|
||||||
@@ -1352,11 +1492,11 @@ end
|
|||||||
--- Register an Enum_ entry in type_name_registry.
|
--- Register an Enum_ entry in type_name_registry.
|
||||||
--- Local helper for parse_typedef_binds. Captures the underlying type (1st arg of `Enum_(<underlying>, <name>)`) and the body fields.
|
--- Local helper for parse_typedef_binds. Captures the underlying type (1st arg of `Enum_(<underlying>, <name>)`) and the body fields.
|
||||||
--- @param underlying string
|
--- @param underlying string
|
||||||
--- @param name string
|
--- @param name string
|
||||||
--- @param body string
|
--- @param body string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
local function register_enum_type(underlying, name, body, pos, line_of, out)
|
local function register_enum_type(underlying, name, body, pos, line_of, out)
|
||||||
local fields = parse_enum_body_fields(body)
|
local fields = parse_enum_body_fields(body)
|
||||||
out.type_name_registry[name] = {
|
out.type_name_registry[name] = {
|
||||||
@@ -1375,18 +1515,18 @@ end
|
|||||||
--- Captures the underlying type ident (LHS of `typedef <type> <alias>;`) and exposes it through the registry.
|
--- Captures the underlying type ident (LHS of `typedef <type> <alias>;`) and exposes it through the registry.
|
||||||
--- The propagation pass follows the underlying_type chain to resolve byte_size.
|
--- The propagation pass follows the underlying_type chain to resolve byte_size.
|
||||||
--- @param underlying string
|
--- @param underlying string
|
||||||
--- @param name string
|
--- @param name string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
local function register_typedef_alias(underlying, name, pos, line_of, out)
|
local function register_typedef_alias(underlying, name, pos, line_of, out)
|
||||||
out.type_name_registry[name] = {
|
out.type_name_registry[name] = {
|
||||||
name = name,
|
name = name,
|
||||||
kind = "typedef",
|
kind = "typedef",
|
||||||
underlying_type = underlying,
|
underlying_type = underlying,
|
||||||
source_line = line_of(pos),
|
source_line = line_of(pos),
|
||||||
source_file = out._source_file,
|
source_file = out._source_file,
|
||||||
pointer_depth = 0,
|
pointer_depth = 0,
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -1396,29 +1536,29 @@ end
|
|||||||
--- 1. `typedef Struct_(<name>) { <body> } <alias>;` adds to type_name_registry (kind="struct").
|
--- 1. `typedef Struct_(<name>) { <body> } <alias>;` adds to type_name_registry (kind="struct").
|
||||||
--- Binds_* aliases also land in out.binds[].
|
--- Binds_* aliases also land in out.binds[].
|
||||||
--- 2. `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
|
--- 2. `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
|
||||||
--- adds to type_name_registry (kind="enum").
|
--- Adds to type_name_registry (kind="enum").
|
||||||
--- 3. `typedef <type> <alias>;` simple typedef alias.
|
--- 3. `typedef <type> <alias>;` simple typedef alias.
|
||||||
--- Adds to type_name_registry (kind="typedef").
|
--- Adds to type_name_registry (kind="typedef").
|
||||||
--- 4. `typedef <type> TSet_(<name>);` duffle TSet_ convention.
|
--- 4. `typedef <type> TSet_(<name>);` duffle TSet_ convention.
|
||||||
--- Strips TSet_ wrapper; adds to type_name_registry (kind="typedef") with underlying_type=<type>.
|
--- Strips TSet_ wrapper; adds to type_name_registry (kind="typedef") with underlying_type=<type>.
|
||||||
---
|
---
|
||||||
--- All four shapes also attach an "unrelated" debug-skip marker (the existing behavior — typedef declarations don't carry atom_dbg_skip).
|
--- All four shapes also attach an "unrelated" debug-skip marker (the existing behavior — typedef declarations don't carry atom_dbg_skip).
|
||||||
--- @param source string
|
--- @param source string
|
||||||
--- @param pos integer
|
--- @param pos integer
|
||||||
--- @param ident_end integer
|
--- @param ident_end integer
|
||||||
--- @param line_of fun(pos: integer): integer
|
--- @param line_of fun(pos: integer): integer
|
||||||
--- @param out SourceScan
|
--- @param out SourceScan
|
||||||
--- @return integer
|
--- @return integer
|
||||||
local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
||||||
local after_typedef = duffle.skip_ws_and_cmt(source, ident_end)
|
local after_typedef = duffle.skip_ws_and_cmt(source, ident_end)
|
||||||
local id2, id2_end = duffle.read_ident(source, after_typedef)
|
local id2, id2_end = duffle.read_ident(source, after_typedef)
|
||||||
if not id2 then return ident_end end
|
if not id2 then return ident_end end
|
||||||
|
|
||||||
-- ── Shape 1: `typedef Struct_(<name>) { <body> } <alias>;` ────────────
|
-- ── Shape 1: `typedef Struct_(<name>) { <body> } <alias>;` ────────────
|
||||||
if id2 == "Struct_" then
|
if id2 == "Struct_" then
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end)
|
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end)
|
||||||
if not inner then return id2_end end
|
if not inner then return id2_end end
|
||||||
local name = duffle.trim(inner)
|
local name = duffle.trim(inner)
|
||||||
|
|
||||||
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
|
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
|
||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
@@ -1428,7 +1568,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
|||||||
|
|
||||||
-- ── Shape 2: `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
|
-- ── Shape 2: `typedef Enum_(<underlying>, <name>) { <body> } <alias>;`
|
||||||
elseif id2 == "Enum_" then
|
elseif id2 == "Enum_" then
|
||||||
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end)
|
local inner, after_paren, open_paren = read_parens_after(source, id2_end, id2_end)
|
||||||
if not inner then return id2_end end
|
if not inner then return id2_end end
|
||||||
-- Split `inner` on the first top-level comma into (<underlying>, <name>).
|
-- Split `inner` on the first top-level comma into (<underlying>, <name>).
|
||||||
local args = duffle.split_top_level_commas(inner)
|
local args = duffle.split_top_level_commas(inner)
|
||||||
@@ -1436,7 +1576,7 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
|||||||
local underlying = duffle.trim(args[1])
|
local underlying = duffle.trim(args[1])
|
||||||
local name = duffle.trim(args[2])
|
local name = duffle.trim(args[2])
|
||||||
|
|
||||||
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
|
local body, after_brace = find_body_braces(source, after_paren, open_paren + 1)
|
||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
register_enum_type(underlying, name, body, pos, line_of, out)
|
register_enum_type(underlying, name, body, pos, line_of, out)
|
||||||
attach_debug_skip_marker(out, "unrelated")
|
attach_debug_skip_marker(out, "unrelated")
|
||||||
@@ -1469,11 +1609,10 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
|||||||
|
|
||||||
-- Shape 4 (TSet_ at id2 position): no preceding underlying span.
|
-- Shape 4 (TSet_ at id2 position): no preceding underlying span.
|
||||||
if id2 == "TSet_" then
|
if id2 == "TSet_" then
|
||||||
local inner, after_paren = read_parens_after(source, id2_end, id2_end)
|
local inner, after_paren = read_parens_after(source, id2_end, id2_end)
|
||||||
if not inner then return id2_end end
|
if not inner then return id2_end end
|
||||||
local tset_name = duffle.trim(inner)
|
local tset_name = duffle.trim(inner)
|
||||||
-- Empty underlying span is acceptable; the TSet_ wrapper itself
|
-- Empty underlying span is acceptable; the TSet_ wrapper itself encodes the alias identity (per the duffle TSet_ convention).
|
||||||
-- encodes the alias identity (per the duffle TSet_ convention).
|
|
||||||
register_typedef_alias("", tset_name, pos, line_of, out)
|
register_typedef_alias("", tset_name, pos, line_of, out)
|
||||||
attach_debug_skip_marker(out, "unrelated")
|
attach_debug_skip_marker(out, "unrelated")
|
||||||
return after_paren
|
return after_paren
|
||||||
@@ -1491,13 +1630,13 @@ local function parse_typedef_binds(source, pos, ident_end, line_of, out)
|
|||||||
while scan < semi_pos do
|
while scan < semi_pos do
|
||||||
scan = duffle.skip_ws_and_cmt(source, scan)
|
scan = duffle.skip_ws_and_cmt(source, scan)
|
||||||
if scan >= semi_pos then break end
|
if scan >= semi_pos then break end
|
||||||
local id, id_end = duffle.read_ident(source, scan)
|
local id, id_end = duffle.read_ident(source, scan)
|
||||||
if not id then
|
if not id then
|
||||||
scan = scan + 1
|
scan = scan + 1
|
||||||
elseif id == "TSet_" then
|
elseif id == "TSet_" then
|
||||||
-- Shape 4 (TSet_ at non-id2 position): grab the parenthesized argument.
|
-- Shape 4 (TSet_ at non-id2 position): grab the parenthesized argument.
|
||||||
local inner, after_paren = read_parens_after(source, id_end, id_end)
|
local inner, after_paren = read_parens_after(source, id_end, id_end)
|
||||||
if inner then
|
if inner then
|
||||||
tset_arg = duffle.trim(inner)
|
tset_arg = duffle.trim(inner)
|
||||||
tset_arg_end = after_paren
|
tset_arg_end = after_paren
|
||||||
tset_pos = scan
|
tset_pos = scan
|
||||||
@@ -1607,6 +1746,16 @@ local function parse_enum_entry(source, body, body_offset, line_of, out, entry_n
|
|||||||
local value, value_end = parse_enum_value(body, after_ws, out)
|
local value, value_end = parse_enum_value(body, after_ws, out)
|
||||||
if value == nil then return value_start end
|
if value == nil then return value_start end
|
||||||
|
|
||||||
|
-- Capture the trailing C-comment (if any) before `skip_ws_and_cmt` discards it.
|
||||||
|
-- The `atom_auto_reg(<scope>, <sym>)` macro expands to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
|
||||||
|
-- so the scope name lives in the comment after the RHS value. Routes through `out.atom_entry_comments`
|
||||||
|
-- for downstream `parse_enum` to split into `out.atom_auto_regs` / `out.phase_auto_regs`.
|
||||||
|
local trailing_cmt = read_trailing_cmt_after(body, value_end)
|
||||||
|
if trailing_cmt then
|
||||||
|
out.atom_entry_comments = out.atom_entry_comments or {}
|
||||||
|
out.atom_entry_comments[entry_name] = trailing_cmt
|
||||||
|
end
|
||||||
|
|
||||||
local after_value = duffle.skip_ws_and_cmt(body, value_end)
|
local after_value = duffle.skip_ws_and_cmt(body, value_end)
|
||||||
local has_atom_reg, end_after_atom_reg = check_bare_atom_reg(body, after_value)
|
local has_atom_reg, end_after_atom_reg = check_bare_atom_reg(body, after_value)
|
||||||
|
|
||||||
@@ -1662,15 +1811,24 @@ local function parse_enum_body(source, body, body_offset, line_of, out)
|
|||||||
else
|
else
|
||||||
local entry_name, name_end = duffle.read_ident(body, pos)
|
local entry_name, name_end = duffle.read_ident(body, pos)
|
||||||
if entry_name then
|
if entry_name then
|
||||||
local after_name = duffle.skip_ws_and_cmt(body, name_end)
|
-- In-enum `atom_auto_reg(<scope>, R_<Sym>)` / `phase_auto_reg(<scope>, R_<Sym>)` markers:
|
||||||
if body:byte(after_name) == BYTE_EQUAL then
|
-- the C preprocessor expands them to `R_<Sym> = R_<Sym>_Code /* atom_auto_reg: <scope> */`,
|
||||||
local new_pos = parse_enum_entry(
|
-- but the metaprogram reads source-as-written so we must dispatch the parser here too.
|
||||||
source, body, body_offset, line_of, out,
|
-- Mirrors the top-level `DECL_PARSERS` entry for `atom_auto_reg` / `phase_auto_reg`.
|
||||||
entry_name, pos, after_name + 1
|
if entry_name == "atom_auto_reg" or entry_name == "phase_auto_reg" then
|
||||||
)
|
local new_pos = parse_auto_reg_marker(body, pos, name_end, line_of, out)
|
||||||
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
|
if new_pos > pos then pos = new_pos else pos = name_end end
|
||||||
else
|
else
|
||||||
pos = name_end
|
local after_name = duffle.skip_ws_and_cmt(body, name_end)
|
||||||
|
if body:byte(after_name) == BYTE_EQUAL then
|
||||||
|
local new_pos = parse_enum_entry(
|
||||||
|
source, body, body_offset, line_of, out,
|
||||||
|
entry_name, pos, after_name + 1
|
||||||
|
)
|
||||||
|
if new_pos > pos then pos = new_pos else pos = after_name + 1 end
|
||||||
|
else
|
||||||
|
pos = name_end
|
||||||
|
end
|
||||||
end
|
end
|
||||||
else
|
else
|
||||||
pos = pos + 1
|
pos = pos + 1
|
||||||
@@ -1700,6 +1858,25 @@ local function parse_enum(source, pos, ident_end, line_of, out)
|
|||||||
if not body then return after_brace end
|
if not body then return after_brace end
|
||||||
parse_enum_body(source, body, body_off, line_of, out)
|
parse_enum_body(source, body, body_off, line_of, out)
|
||||||
|
|
||||||
|
-- Route `atom_auto_reg:` / `phase_auto_reg:` markers discovered in trailing C-comments
|
||||||
|
-- into the per-source `atom_auto_regs` / `phase_auto_regs` projections.
|
||||||
|
-- Pattern matches the RHS expansion `R_<Sym> = R_<Sym>_Code /* <kind>_auto_reg: <scope> */`
|
||||||
|
-- emitted by the `atom_auto_reg` / `phase_auto_reg` macros in dsl.atom.h.
|
||||||
|
for entry_name, cmt_text in pairs(out.atom_entry_comments or {}) do
|
||||||
|
local atom_scope = cmt_text:match("atom_auto_reg:%s*([%w_]+)")
|
||||||
|
if atom_scope then
|
||||||
|
out.atom_auto_regs = out.atom_auto_regs or {}
|
||||||
|
out.atom_auto_regs[atom_scope] = out.atom_auto_regs[atom_scope] or {}
|
||||||
|
out.atom_auto_regs[atom_scope][entry_name] = entry_name
|
||||||
|
end
|
||||||
|
local phase_scope = cmt_text:match("phase_auto_reg:%s*([%w_]+)")
|
||||||
|
if phase_scope then
|
||||||
|
out.phase_auto_regs = out.phase_auto_regs or {}
|
||||||
|
out.phase_auto_regs[phase_scope] = out.phase_auto_regs[phase_scope] or {}
|
||||||
|
out.phase_auto_regs[phase_scope][entry_name] = entry_name
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
return after_brace
|
return after_brace
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -1713,12 +1890,18 @@ end
|
|||||||
|
|
||||||
local DECL_PARSERS = {
|
local DECL_PARSERS = {
|
||||||
MipsAtom_ = parse_mips_atom,
|
MipsAtom_ = parse_mips_atom,
|
||||||
|
MipsAtom_Proc_ = parse_mips_atom_proc,
|
||||||
MipsAtomComp_ = parse_mips_atom_comp,
|
MipsAtomComp_ = parse_mips_atom_comp,
|
||||||
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
|
MipsAtomComp_Proc_ = parse_mips_atom_comp_proc,
|
||||||
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
|
-- `atom_dbg_skip` is the only debug-skip parser entry. Every other
|
||||||
-- identifier follows the ordinary unrelated-token path; there is no alias.
|
-- identifier follows the ordinary unrelated-token path; there is no alias.
|
||||||
atom_dbg_skip = parse_dbg_skip_marker,
|
atom_dbg_skip = parse_dbg_skip_marker,
|
||||||
atom_dbg_reg_default = parse_atom_dbg_reg_default,
|
atom_dbg_reg_default = parse_atom_dbg_reg_default,
|
||||||
|
-- `atom_auto_reg(atom, R_<Sym>)` and `phase_auto_reg(phase, R_<Sym>)` populate per-source
|
||||||
|
-- `out.atom_auto_regs` / `out.phase_auto_regs`; the cross-source merge lands in
|
||||||
|
-- `corpus.atom_auto_regs` / `corpus.phase_auto_regs` (first-wins).
|
||||||
|
atom_auto_reg = parse_auto_reg_marker,
|
||||||
|
phase_auto_reg = parse_auto_reg_marker,
|
||||||
MipsCode = parse_mips_code,
|
MipsCode = parse_mips_code,
|
||||||
typedef = parse_typedef_binds,
|
typedef = parse_typedef_binds,
|
||||||
_Pragma = parse_pragma_macro,
|
_Pragma = parse_pragma_macro,
|
||||||
@@ -1753,6 +1936,14 @@ local function scan_source(source, source_file, code_macros, code_macro_bodies)
|
|||||||
debug_skip_markers = {},
|
debug_skip_markers = {},
|
||||||
types = {},
|
types = {},
|
||||||
atom_views = {},
|
atom_views = {},
|
||||||
|
-- Per-source projection for `atom_auto_reg(<atom>, R_<Sym>)` markers.
|
||||||
|
-- Each entry is keyed by atom_name; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
|
||||||
|
-- Merged cross-source into `corpus.atom_auto_regs` (first-wins).
|
||||||
|
atom_auto_regs = {},
|
||||||
|
-- Per-source projection for `phase_auto_reg(<phase>, R_<Sym>)` markers.
|
||||||
|
-- Each entry is keyed by phase_label; the inner table maps `R_<Sym>` -> `R_<Sym>` (raw LHS sym).
|
||||||
|
-- Merged cross-source into `corpus.phase_auto_regs` (first-wins).
|
||||||
|
phase_auto_regs = {},
|
||||||
line_of = line_of,
|
line_of = line_of,
|
||||||
-- Source-derived register-alias registry (atom_reg opt-in entries).
|
-- Source-derived register-alias registry (atom_reg opt-in entries).
|
||||||
-- Keys are full R_* idents (never stripped); see parse_enum / parse_enum_body.
|
-- Keys are full R_* idents (never stripped); see parse_enum / parse_enum_body.
|
||||||
@@ -1957,7 +2148,7 @@ local function merge_named_with_sites(registry, name, new_entry, site, collision
|
|||||||
registry[name].sites = { site }
|
registry[name].sites = { site }
|
||||||
return
|
return
|
||||||
end
|
end
|
||||||
local existing = registry[name]
|
local existing = registry[name]
|
||||||
local new_shape = shape_fn(new_entry)
|
local new_shape = shape_fn(new_entry)
|
||||||
local old_shape = shape_fn(existing)
|
local old_shape = shape_fn(existing)
|
||||||
if new_shape == old_shape and new_shape ~= "" then
|
if new_shape == old_shape and new_shape ~= "" then
|
||||||
@@ -1992,6 +2183,8 @@ local function merge_corpus_registries(corpus)
|
|||||||
corpus.atom_ctxs = corpus.atom_ctxs or {}
|
corpus.atom_ctxs = corpus.atom_ctxs or {}
|
||||||
corpus.atom_phases = corpus.atom_phases or {}
|
corpus.atom_phases = corpus.atom_phases or {}
|
||||||
corpus.atom_infos = corpus.atom_infos or {}
|
corpus.atom_infos = corpus.atom_infos or {}
|
||||||
|
corpus.atom_auto_regs = corpus.atom_auto_regs or {}
|
||||||
|
corpus.phase_auto_regs = corpus.phase_auto_regs or {}
|
||||||
corpus.collisions = corpus.collisions or {}
|
corpus.collisions = corpus.collisions or {}
|
||||||
|
|
||||||
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
-- Replace the existing corpus collections with empty tables so a re-run on the same corpus produces identical state (deterministic merge).
|
||||||
@@ -2035,7 +2228,7 @@ local function merge_corpus_registries(corpus)
|
|||||||
corpus.collisions, "binds", bind_shape)
|
corpus.collisions, "binds", bind_shape)
|
||||||
end
|
end
|
||||||
|
|
||||||
-- atoms_by_name: MipsAtom_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
|
-- atoms_by_name: MipsAtom_(name) + MipsAtom_Proc_(name) + MipsAtomComp_(name) + MipsAtomComp_Proc_(name).
|
||||||
-- Each atom carries `{line, name, body, body_off, kind, raw_name, ...}`.
|
-- Each atom carries `{line, name, body, body_off, kind, raw_name, ...}`.
|
||||||
-- Duplicate atom names across sources are first-wins + collision; see the atom_infos block below for the evidence list.
|
-- Duplicate atom names across sources are first-wins + collision; see the atom_infos block below for the evidence list.
|
||||||
for _, atom_entry in ipairs(scan.atoms or {}) do
|
for _, atom_entry in ipairs(scan.atoms or {}) do
|
||||||
@@ -2070,6 +2263,22 @@ local function merge_corpus_registries(corpus)
|
|||||||
corpus.collisions, "phase", phase_shape)
|
corpus.collisions, "phase", phase_shape)
|
||||||
end
|
end
|
||||||
|
|
||||||
|
-- atom_auto_regs: keyed by atom scope name; each carries a `{R_<Sym> = R_<Sym>}` map.
|
||||||
|
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
|
||||||
|
for atom_scope, syms in pairs(scan.atom_auto_regs or {}) do
|
||||||
|
if corpus.atom_auto_regs[atom_scope] == nil then
|
||||||
|
corpus.atom_auto_regs[atom_scope] = syms
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- phase_auto_regs: keyed by phase label; each carries a `{R_<Sym> = R_<Sym>}` map.
|
||||||
|
-- Per-source entries are simple inner maps (no body / no shape comparison); first-wins suffices.
|
||||||
|
for phase_label, syms in pairs(scan.phase_auto_regs or {}) do
|
||||||
|
if corpus.phase_auto_regs[phase_label] == nil then
|
||||||
|
corpus.phase_auto_regs[phase_label] = syms
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
-- atom_infos: ALWAYS append every record in source/declaration order.
|
-- atom_infos: ALWAYS append every record in source/declaration order.
|
||||||
-- Duplicates are preserved so the annotation pass can flag them via `check_unique_annotation`;
|
-- Duplicates are preserved so the annotation pass can flag them via `check_unique_annotation`;
|
||||||
-- The merge is purely order-preserving.
|
-- The merge is purely order-preserving.
|
||||||
|
|||||||
+492
-162
@@ -1,20 +1,17 @@
|
|||||||
--- passes/static_analysis.lua — Per-atom static-analysis checks.
|
--- passes/static_analysis.lua — Per-atom static-analysis checks.
|
||||||
---
|
--- Ownership: `ctx.shared.corpus` canonical merged registry; per-source fallback synthesis is rejected.
|
||||||
--- Ownership: `ctx.shared.corpus` is the canonical merged registry; per-source fallback synthesis is rejected.
|
|
||||||
--- `atom.paths` supplies the emitted and analysis projections consumed by this pass.
|
--- `atom.paths` supplies the emitted and analysis projections consumed by this pass.
|
||||||
---
|
---
|
||||||
--- Per-atom rules:
|
--- Per-atom rules:
|
||||||
--- 1. transfer_hazards: A single forward walker (`analyze_hardware_relations`) reads `atom.paths.word_events` once per atom.
|
--- 1. transfer_hazards: A single forward walker (`analyze_hardware_relations`) reads `atom.paths.word_events` once per atom.
|
||||||
--- For each emitted word event it (a) inspects pending CPU/COP0/COP2/GTE relations against the event as CONSUMER
|
--- For each emitted word event it (a) inspects pending CPU / COP0 / COP2 / GTE relations against the event as CONSUMER
|
||||||
--- (recording a hazard on `atom.paths.hazards` when the producer→consumer gap is below the required retire-slot count),
|
--- (recording a hazard on `atom.paths.hazards` when the producer→consumer gap is below the required retire-slot count),
|
||||||
--- (b) applies the event's GPR value effects (`duffle.INSTRUCTION_GPR_EFFECTS`) to `atom.paths.forward_state.gpr_values`,
|
--- (b) Applies the event's GPR value effects (`duffle.INSTRUCTION_GPR_EFFECTS`) to `atom.paths.forward_state.gpr_values`,
|
||||||
--- applies bounded constant propagation, and stages matching relation rows as PRODUCERS (with `destination_match` filters, e.g. for the IRGB fan-out).
|
--- applies bounded constant propagation, and stages matching relation rows as PRODUCERS (with `destination_match` filters, e.g. for the IRGB fan-out).
|
||||||
--- The `transfer_hazards` CHECK_RULES reader projects `atom.paths.hazards` into per-atom findings.
|
--- The `transfer_hazards` CHECK_RULES reader projects `atom.paths.hazards` into per-atom findings.
|
||||||
--- The reader does NOT re-walk source; this is the per-check purity contract.
|
--- The reader does NOT re-walk source. The walker runs once per atom before the per-atom dispatch; the reader runs inside the same dispatch.
|
||||||
--- The walker runs once per atom before the per-atom dispatch; the reader runs inside the same dispatch.
|
|
||||||
--- 2. control_transfer_delay_slot_use: For every emitted branch/jump/call encoder in `duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES`
|
--- 2. control_transfer_delay_slot_use: For every emitted branch/jump/call encoder in `duffle.CONTROL_TRANSFER_DELAY_SLOT_POLICIES`
|
||||||
--- (the six `branch_*` encoders plus `jump` / `jump_reg` / `jump_link` / `call_reg` / `call_addr`),
|
--- (the six `branch_*` encoders plus `jump` / `jump_reg` / `jump_link` / `call_reg` / `call_addr`), inspect the next emitted event in `atom.paths.word_events`.
|
||||||
--- inspect the next emitted event in `atom.paths.word_events`.
|
|
||||||
--- Emit an `info`-severity finding when the successor is `nop` or absent (the next emitted word IS the hardware delay slot).
|
--- Emit an `info`-severity finding when the successor is `nop` or absent (the next emitted word IS the hardware delay slot).
|
||||||
--- `jump_reg(R_AtomJmp)` is suppressed by policy (the fixed `mac_yield()` handshake).
|
--- `jump_reg(R_AtomJmp)` is suppressed by policy (the fixed `mac_yield()` handshake).
|
||||||
--- `nop2` needs no special case: emission-model emits two `nop` events for it, so the first expansion is the hardware delay slot.
|
--- `nop2` needs no special case: emission-model emits two `nop` events for it, so the first expansion is the hardware delay slot.
|
||||||
@@ -42,8 +39,8 @@
|
|||||||
--- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts.
|
--- `── Info` section renders finding-level info between `── Warnings` and the per-atom cycle counts.
|
||||||
---
|
---
|
||||||
--- The structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`, `control_transfer_delay_slot_use`) skip atoms/components with `debug_skip == true`.
|
--- The structural handshake checks (`mac_yield_uniformity`, `hazard_nop_use`, `control_transfer_delay_slot_use`) skip atoms/components with `debug_skip == true`.
|
||||||
--- The `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
|
--- `atom_dbg_skip` marker designates runtime-helper declarations whose structure is fixed by the tape runtime (e.g. `tape_exit`, `ac_yield`).
|
||||||
--- Flagging them as "missing mac_yield" or "BD slot is redundant" is signal noise, not a logic failure.
|
--- Flagging them as "missing mac_yield" or "BD slot is redundant".
|
||||||
--- Other checks (transfer_hazards, gpu_portstore_shape, abi_handoff, enum_alias_membership, …) still apply to debug_skip declarations because real hazards / typos can still surface in them.
|
--- Other checks (transfer_hazards, gpu_portstore_shape, abi_handoff, enum_alias_membership, …) still apply to debug_skip declarations because real hazards / typos can still surface in them.
|
||||||
---
|
---
|
||||||
--- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table:
|
--- The orchestrator (`ps1_meta.lua`) wires this module in via the PASSES table:
|
||||||
@@ -101,10 +98,10 @@ local OUTPUT_EXTENSION = ".static_analysis.txt"
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field sources SourceFile[]
|
--- @field sources SourceFile[]
|
||||||
@@ -121,43 +118,42 @@ local OUTPUT_EXTENSION = ".static_analysis.txt"
|
|||||||
--- @field outputs table[]
|
--- @field outputs table[]
|
||||||
--- @field errors table[]
|
--- @field errors table[]
|
||||||
--- @field warnings table[]
|
--- @field warnings table[]
|
||||||
--- @field info table[] -- finding-level info (kind == "info"); distinct from per-source scanned/cycles summary rows
|
--- @field info table[] -- Finding-level info (kind == "info"); distinct from per-source scanned/cycles summary rows
|
||||||
|
|
||||||
--- @alias AtomName string -- lower_snake_case atom nameMacroName string -- lower_snake_case macro identifier
|
--- @alias AtomName string -- lower_snake_case atom nameMacroName string -- lower_snake_case macro identifier
|
||||||
--- @alias CheckName string -- "transfer_hazards" | "control_transfer_delay_slot_use" | "mac_yield_uniformity" | "yield_load_tail_pairing" | "abi_handoff" | "gpu_portstore_shape" | "per_atom_cycle_budget" | "enum_alias_membership" | "atom_type_consistency" | "binds_no_substruct_deref"
|
--- @alias CheckName string -- "transfer_hazards" | "control_transfer_delay_slot_use" | "mac_yield_uniformity" | "yield_load_tail_pairing" | "abi_handoff" | "gpu_portstore_shape" | "per_atom_cycle_budget" | "enum_alias_membership" | "atom_type_consistency" | "binds_no_substruct_deref"
|
||||||
|
|
||||||
--- @class AtomBody
|
--- @class AtomBody
|
||||||
--- @field line integer -- source line of the atom declaration
|
--- @field line integer -- Source line of the atom declaration
|
||||||
--- @field name AtomName -- atom name (e.g. "cube_g4_face")
|
--- @field name AtomName -- Atom name (e.g. "cube_g4_face")
|
||||||
--- @field body string -- the brace-delimited body (without the braces)
|
--- @field body string -- Brace-delimited body (without the braces)
|
||||||
--- @field body_off integer -- char offset of body[1] in source
|
--- @field body_off integer -- char offset of body[1] in source
|
||||||
--- @field kind string -- "atom" | "comp_bare" | "comp_proc"
|
--- @field kind string -- "atom" | "comp_bare" | "comp_proc"
|
||||||
|
|
||||||
--- @class Token
|
--- @class Token
|
||||||
--- @field tok string -- the raw token text (trimmed)
|
--- @field tok string -- Raw token text (trimmed)
|
||||||
--- @field line integer -- source line of the token's start
|
--- @field line integer -- Source line of the token's start
|
||||||
--- @field ident string|nil -- the leading ident of the token (if any)
|
--- @field ident string|nil -- Leading ident of the token (if any)
|
||||||
--- @field kind string -- "n_words" | "mac_yield" | "gte_cmdw" | "mac_format" | "mac_gte_store" | "mac_insert_ot_tag" | "atom_label" | "atom_offset" | "other"
|
--- @field kind string -- "n_words" | "mac_yield" | "gte_cmdw" | "mac_format" | "mac_gte_store" | "mac_insert_ot_tag" | "atom_label" | "atom_offset" | "other"
|
||||||
|
|
||||||
--- @class Finding
|
--- @class Finding
|
||||||
--- @field line integer -- source line of the finding
|
--- @field line integer -- Source line of the finding
|
||||||
--- @field atom AtomName -- the atom this finding is for (or "")
|
--- @field atom AtomName -- Atom this finding is for (or "")
|
||||||
--- @field check CheckName -- the check identifier
|
--- @field check CheckName -- Check identifier
|
||||||
--- @field kind string -- "error" | "warning" | "info"
|
--- @field kind string -- "error" | "warning" | "info"
|
||||||
--- @field msg string -- the finding message
|
--- @field msg string -- Finding message
|
||||||
|
|
||||||
--- @class AtomAnalysis
|
--- @class AtomAnalysis
|
||||||
--- @field atom AtomBody
|
--- @field atom AtomBody
|
||||||
--- @field tokens Token[] -- the tokens in the atom body, annotated
|
--- @field tokens Token[] -- Tokens in the atom body, annotated
|
||||||
--- @field findings Finding[] -- findings for this atom
|
--- @field findings Finding[] -- Findings for this atom
|
||||||
--- @field total_cycles integer -- sum of token cycle costs
|
--- @field total_cycles integer -- Sum of token cycle costs
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- Per-word-event helpers
|
-- Per-word-event helpers
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
-- Pick the source-line field that best represents "where in the user's source file is this word?".
|
-- Pick the source-line field that best represents "where in the user's source file is this word?".
|
||||||
--
|
|
||||||
-- `word_events` (populated by `passes/emission_model.lua::stamp_root_provenance`) carry four line fields:
|
-- `word_events` (populated by `passes/emission_model.lua::stamp_root_provenance`) carry four line fields:
|
||||||
-- * `call_line` — physical line in the ROOT atom's source (the line of the `mac_X(...)` call site that triggered this emission, or `body_line` for direct words in the atom body)
|
-- * `call_line` — physical line in the ROOT atom's source (the line of the `mac_X(...)` call site that triggered this emission, or `body_line` for direct words in the atom body)
|
||||||
-- * `body_line` — physical line in the body containing the emitted word (the atom body for direct words; the component body for words expanded inside `mac_X(...)`)
|
-- * `body_line` — physical line in the body containing the emitted word (the atom body for direct words; the component body for words expanded inside `mac_X(...)`)
|
||||||
@@ -169,8 +165,7 @@ local OUTPUT_EXTENSION = ".static_analysis.txt"
|
|||||||
-- The user editing their atom body expects the line to point at THEIR source — i.e. the line where `mac_yield()`
|
-- The user editing their atom body expects the line to point at THEIR source — i.e. the line where `mac_yield()`
|
||||||
-- was called (e.g. `hello_gte_tape.c:35`). That line is `call_line`.
|
-- was called (e.g. `hello_gte_tape.c:35`). That line is `call_line`.
|
||||||
--
|
--
|
||||||
-- For direct words in the atom body (no invocation wrapping them), `call_line == body_line` already,
|
-- For direct words in the atom body (no invocation wrapping them), `call_line == body_line` already, so `call_line` works for both cases.
|
||||||
-- so `call_line` works for both cases.
|
|
||||||
local function line_for_word_event(ev)
|
local function line_for_word_event(ev)
|
||||||
if ev == nil then return 0 end
|
if ev == nil then return 0 end
|
||||||
return ev.call_line or ev.body_line or ev.line or ev.def_line or 0
|
return ev.call_line or ev.body_line or ev.line or ev.def_line or 0
|
||||||
@@ -193,46 +188,44 @@ end
|
|||||||
--
|
--
|
||||||
-- The classification is stored on `atom.paths.tok_class` as an array indexed by token index (1..#tokens).
|
-- The classification is stored on `atom.paths.tok_class` as an array indexed by token index (1..#tokens).
|
||||||
-- Each entry has:
|
-- Each entry has:
|
||||||
-- ident — the leading identifier (e.g. "load_word", "gte_cmdw_rtpt", "nop", "mac_yield")
|
-- ident — the leading identifier (e.g. "load_word", "gte_cmdw_rtpt", "nop", "mac_yield")
|
||||||
-- nop_words — 0 / 1 / 2 (for "nop" / "nop2" / anything else)
|
-- nop_words — 0 / 1 / 2 (for "nop" / "nop2" / anything else)
|
||||||
-- nop_prefix — consecutive nop words ending just BEFORE this token (forward-pass pre-compute;
|
-- nop_prefix — consecutive nop words ending just BEFORE this token (forward-pass pre-compute; makes preceding-nop lookup O(N))
|
||||||
-- makes preceding-nop lookup O(N))
|
-- is_yield — true if this token is `mac_yield` or `mac_yield(...)`
|
||||||
-- is_yield — true if this token is `mac_yield` or `mac_yield(...)`
|
-- is_atom_label — true if this token is `atom_label(name)`; label_name has the name
|
||||||
-- is_atom_label — true if this token is `atom_label(name)`; label_name has the name
|
-- is_branch — true if this token is `branch_*(...)` OR an unconditional-jump-with-offset (`jump(off)` / `call_addr(off)`); branch_label has the target label or false
|
||||||
-- is_branch — true if this token is `branch_*(...)` OR an unconditional-jump-with-offset (`jump(off)` / `call_addr(off)`); branch_label has the target label or false
|
|
||||||
-- is_unconditional_jump — true if this token is `jump` or `call_addr` (BD slot + single successor — taken only; no fall-through).
|
-- is_unconditional_jump — true if this token is `jump` or `call_addr` (BD slot + single successor — taken only; no fall-through).
|
||||||
-- Mutually exclusive with the conditional-branch semantics; combined with `is_branch` above.
|
-- Mutually exclusive with the conditional-branch semantics; combined with `is_branch` above.
|
||||||
-- is_terminal_jump — true if this token is `jump_reg` / `call_reg` / `jump_link` (transfers control OUT of the current atom; the `mac_yield()` handshake ends in `jump_reg(R_AtomJmp), nop`).
|
-- is_terminal_jump — true if this token is `jump_reg` / `call_reg` / `jump_link` (transfers control OUT of the current atom; the `mac_yield()` handshake ends in `jump_reg(R_AtomJmp), nop`).
|
||||||
-- No offset field — `atom_offset` is invalid here. Terminates the current path in the CFG.
|
-- No offset field — `atom_offset` is invalid here. Terminates the current path in the CFG.
|
||||||
-- is_load — true if this token starts with any of: load_word, load_half, load_half_u, load_byte,
|
-- is_load — true if this token starts with any of: load_word, load_half, load_half_u, load_byte,
|
||||||
-- load_byte_u, gte_lw, gte_lwc2. These all have MIPS load-delay semantics (the
|
-- load_byte_u, gte_lw, gte_lwc2. These all have MIPS load-delay semantics (the destination register is volatile for 1 word after the load).
|
||||||
-- destination register is volatile for 1 word after the load).
|
-- is_store_word — true if this token starts with `store_word(`
|
||||||
-- is_store_word — true if this token starts with `store_word(`
|
|
||||||
--
|
--
|
||||||
-- Checks that need the leading ident use `tok_class.ident` instead of re-matching the token string.
|
-- Checks that need the leading ident use `tok_class.ident` instead of re-matching the token string.
|
||||||
-- Checks that need "how many nops before token i" use `tok_class.nop_prefix` instead of walking backwards.
|
-- Checks that need "how many nops before token i" use `tok_class.nop_prefix` instead of walking backwards.
|
||||||
|
|
||||||
--- @class TokClass
|
--- @class TokClass
|
||||||
--- @field ident string -- leading identifier
|
--- @field ident string -- lLading identifier
|
||||||
--- @field nop_words integer -- 0/1/2
|
--- @field nop_words integer -- 0 / 1 / 2
|
||||||
--- @field nop_prefix integer -- consecutive nop words before this token
|
--- @field nop_prefix integer -- Consecutive nop words before this token
|
||||||
--- @field is_yield boolean
|
--- @field is_yield boolean
|
||||||
--- @field is_atom_label boolean
|
--- @field is_atom_label boolean
|
||||||
--- @field label_name string|nil -- for atom_label(name)
|
--- @field label_name string|nil -- For atom_label(name)
|
||||||
--- @field is_branch boolean -- conditional branch OR unconditional-jump-with-offset
|
--- @field is_branch boolean -- Conditional branch OR unconditional-jump-with-offset
|
||||||
--- @field is_unconditional_jump boolean -- `jump` / `call_addr` only
|
--- @field is_unconditional_jump boolean -- `jump` / `call_addr` only
|
||||||
--- @field is_terminal_jump boolean -- `jump_reg` / `call_reg` / `jump_link` only
|
--- @field is_terminal_jump boolean -- `jump_reg` / `call_reg` / `jump_link` only
|
||||||
--- @field branch_label string|false|nil -- for branch_*(..., atom_offset(F, label)) OR jump/call_addr
|
--- @field branch_label string|false|nil -- For branch_*(..., atom_offset(F, label)) OR jump/call_addr
|
||||||
--- @field is_load boolean -- load_word | load_half | load_half_u | load_byte | load_byte_u | gte_lw | gte_lwc2
|
--- @field is_load boolean -- load_word | load_half | load_half_u | load_byte | load_byte_u | gte_lw | gte_lwc2
|
||||||
--- @field is_store_word boolean
|
--- @field is_store_word boolean
|
||||||
--- @field mac_format_shape string|nil -- "f3" / "g4" etc. for mac_format_X_color; nil otherwise
|
--- @field mac_format_shape string|nil -- "f3" / "g4" etc. for mac_format_X_color; nil otherwise
|
||||||
--- @field is_gte_store boolean -- ident matches `mac_gte_store_<shape>`
|
--- @field is_gte_store boolean -- Ident matches `mac_gte_store_<shape>`
|
||||||
--- @field is_ot_tag boolean -- ident matches `mac_insert_ot_tag_<shape>`
|
--- @field is_ot_tag boolean -- Ident matches `mac_insert_ot_tag_<shape>`
|
||||||
--- @field writes_r_prim_cursor boolean -- store_word targeting R_PrimCursor
|
--- @field writes_r_prim_cursor boolean -- store_word targeting R_PrimCursor
|
||||||
--- @field reads_r_tape_ptr boolean -- any token referencing R_TapePtr
|
--- @field reads_r_tape_ptr boolean -- Any token referencing R_TapePtr
|
||||||
--- @field o_arg1 string|nil -- first arg of O_(<a>, <b>) captures; nil for non-O_ tokens
|
--- @field o_arg1 string|nil -- First arg of O_(<a>, <b>) captures; nil for non-O_ tokens
|
||||||
--- @field o_arg2 string|nil -- second arg of O_(<a>, <b>) captures
|
--- @field o_arg2 string|nil -- Second arg of O_(<a>, <b>) captures
|
||||||
--- @field s_arg1 string|nil -- arg of S_(<a>) captures; nil for non-S_ tokens
|
--- @field s_arg1 string|nil -- Arg of S_(<a>) captures; nil for non-S_ tokens
|
||||||
|
|
||||||
-- The set of MIPS instruction idents that have a load-delay slot.
|
-- The set of MIPS instruction idents that have a load-delay slot.
|
||||||
-- Per MIPS I R3000A: `lw`, `lh`, `lhu`, `lb`, `lbu`, `lwc2` (gte_lw).
|
-- Per MIPS I R3000A: `lw`, `lh`, `lhu`, `lb`, `lbu`, `lwc2` (gte_lw).
|
||||||
@@ -263,8 +256,21 @@ local BRANCH_PATTERN = "^branch_[%w_]+%s*%("
|
|||||||
-- The C preprocessor expands it BEFORE the metaprogram sees the source, but for source-level metadata consistency we still match it here and classify it as a branch_equal.
|
-- The C preprocessor expands it BEFORE the metaprogram sees the source, but for source-level metadata consistency we still match it here and classify it as a branch_equal.
|
||||||
-- This keeps `consuming_encoder` canonical for any downstream tooling that consults the metadata field.
|
-- This keeps `consuming_encoder` canonical for any downstream tooling that consults the metadata field.
|
||||||
local JUMP_REL_PATTERN = "^jump_rel%s*%("
|
local JUMP_REL_PATTERN = "^jump_rel%s*%("
|
||||||
local UNCOND_JUMP_PATTERN = "^%f[%w](jump|call_addr)%f[%W]"
|
local UNCOND_JUMP_PATTERNS = {
|
||||||
local TERMINAL_JUMP_PATTERN = "^%f[%w](jump_reg|call_reg|jump_link)%f[%W]"
|
"^%f[%w]jump%f[%W]",
|
||||||
|
"^%f[%w]call_addr%f[%W]",
|
||||||
|
}
|
||||||
|
local TERMINAL_JUMP_PATTERNS = {
|
||||||
|
"^%f[%w]jump_reg%f[%W]",
|
||||||
|
"^%f[%w]call_reg%f[%W]",
|
||||||
|
"^%f[%w]jump_link%f[%W]",
|
||||||
|
}
|
||||||
|
local function matches_any(tok, patterns)
|
||||||
|
for i = 1, #patterns do
|
||||||
|
if tok:match(patterns[i]) then return true end
|
||||||
|
end
|
||||||
|
return false
|
||||||
|
end
|
||||||
|
|
||||||
local function classify_tokens(tokens)
|
local function classify_tokens(tokens)
|
||||||
local n = #tokens
|
local n = #tokens
|
||||||
@@ -308,13 +314,13 @@ local function classify_tokens(tokens)
|
|||||||
-- Both encode a 16-bit signed relative word offset.
|
-- Both encode a 16-bit signed relative word offset.
|
||||||
is_branch = true
|
is_branch = true
|
||||||
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
|
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
|
||||||
elseif tok:match(UNCOND_JUMP_PATTERN) then
|
elseif matches_any(tok, UNCOND_JUMP_PATTERNS) then
|
||||||
-- Unconditional absolute jump / call: `jump(off)` / `call_addr(off)`.
|
-- Unconditional absolute jump / call: `jump(off)` / `call_addr(off)`.
|
||||||
-- One immediate offset field; can carry an `atom_offset(F, T)` marker (the offsets pass dispatches on `consuming_encoder` — see `passes/offsets.lua::compute_offsets`).
|
-- One immediate offset field; can carry an `atom_offset(F, T)` marker (the offsets pass dispatches on `consuming_encoder` — see `passes/offsets.lua::compute_offsets`).
|
||||||
is_branch = true
|
is_branch = true
|
||||||
is_unconditional_jump = true
|
is_unconditional_jump = true
|
||||||
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
|
branch_label = tok:match("atom_offset%s*%([^,]+,%s*([%w_]+)%s*%)") or false
|
||||||
elseif tok:match(TERMINAL_JUMP_PATTERN) then
|
elseif matches_any(tok, TERMINAL_JUMP_PATTERNS) then
|
||||||
-- Register-form jump / call: no offset field; `atom_offset` is invalid here (the offsets pass will error if one is supplied).
|
-- Register-form jump / call: no offset field; `atom_offset` is invalid here (the offsets pass will error if one is supplied).
|
||||||
-- Transfers control OUT of the current atom — the CFG treats this as a path terminator.
|
-- Transfers control OUT of the current atom — the CFG treats this as a path terminator.
|
||||||
is_terminal_jump = true
|
is_terminal_jump = true
|
||||||
@@ -343,7 +349,7 @@ local function classify_tokens(tokens)
|
|||||||
is_atom_label = is_atom_label,
|
is_atom_label = is_atom_label,
|
||||||
label_name = label_name,
|
label_name = label_name,
|
||||||
is_branch = is_branch,
|
is_branch = is_branch,
|
||||||
is_unconditional_jump = is_unconditional_jump,
|
is_unconditional_jump = is_unconditional_jump,
|
||||||
is_terminal_jump = is_terminal_jump,
|
is_terminal_jump = is_terminal_jump,
|
||||||
branch_label = branch_label,
|
branch_label = branch_label,
|
||||||
is_load = is_load,
|
is_load = is_load,
|
||||||
@@ -401,15 +407,13 @@ end
|
|||||||
-- therefore counts ONLY words strictly between the producer and the consumer.
|
-- therefore counts ONLY words strictly between the producer and the consumer.
|
||||||
-- ─────────────────────────────────────────────────────────────────────────
|
-- ─────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
-- True iff `consumer_event` is a GTE command (gte_cmdw_* or one of the human-readable aliases
|
-- True iff `consumer_event` is a GTE command (gte_cmdw_* or one of the human-readable aliases mapped in `duffle.GTE_COMMAND_ALIASES`).
|
||||||
-- mapped in `duffle.GTE_COMMAND_ALIASES`). Used by the LWC2 retirement-regime dispatch in the
|
-- Used by the LWC2 retirement-regime dispatch in the forward walker: a GTE-command consumer can read the LWC2 result in the very next slot
|
||||||
-- forward walker: a GTE-command consumer can read the LWC2 result in the very next slot (the GTE
|
-- (the GTE pipeline latches the LWC2 data); any other consumer must observe the standard MIPS load delay (gap >= 1).
|
||||||
-- pipeline latches the LWC2 data); any other consumer must observe the standard MIPS load delay
|
|
||||||
-- (gap >= 1).
|
|
||||||
local function is_gte_command(consumer_event)
|
local function is_gte_command(consumer_event)
|
||||||
local tok = consumer_event.encoder or consumer_event.ident or ""
|
local tok = consumer_event.encoder or consumer_event.ident or ""
|
||||||
if tok:sub(1, 9) == "gte_cmdw_" then return true end
|
if tok:sub(1, 9) == "gte_cmdw_" then return true end
|
||||||
local aliases = duffle.GTE_COMMAND_ALIASES or {}
|
local aliases = duffle.GTE_COMMAND_ALIASES or {}
|
||||||
return aliases[tok] ~= nil
|
return aliases[tok] ~= nil
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -422,7 +426,7 @@ local function is_cop2_consumer_of(consumer_event, destination, producer_rel)
|
|||||||
-- not used by MTC2/CTC2 today because the "consumer" is a GTE command and its reads are not operand positions.)
|
-- not used by MTC2/CTC2 today because the "consumer" is a GTE command and its reads are not operand positions.)
|
||||||
local args = consumer_event.args or {}
|
local args = consumer_event.args or {}
|
||||||
for _, pos in ipairs(args) do
|
for _, pos in ipairs(args) do
|
||||||
if pos == destination then return true end
|
if pos == destination then return true end
|
||||||
end
|
end
|
||||||
-- Match via the command's input set: the consumer encoder resolves to a `gte_cmdw_*`
|
-- Match via the command's input set: the consumer encoder resolves to a `gte_cmdw_*`
|
||||||
-- short form whose `duffle.GTE_COMMAND_INPUTS` entry includes the destination (or a fan-out target).
|
-- short form whose `duffle.GTE_COMMAND_INPUTS` entry includes the destination (or a fan-out target).
|
||||||
@@ -448,7 +452,7 @@ local function is_cop2_consumer_of(consumer_event, destination, producer_rel)
|
|||||||
end
|
end
|
||||||
|
|
||||||
-- True iff `consumer_event` reads the GPR operand at any position the destination register occupies.
|
-- True iff `consumer_event` reads the GPR operand at any position the destination register occupies.
|
||||||
-- The read-position lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
|
-- read_pos lookup consults `duffle.OPERAND_READ_POSITIONS` for the consumer's encoder and walks each `args[pos]` to find an operand-equal match.
|
||||||
local function is_gpr_consumer_of(consumer_event, destination)
|
local function is_gpr_consumer_of(consumer_event, destination)
|
||||||
local consumer_token = consumer_event.encoder or consumer_event.ident
|
local consumer_token = consumer_event.encoder or consumer_event.ident
|
||||||
local read_pos = duffle.OPERAND_READ_POSITIONS or {}
|
local read_pos = duffle.OPERAND_READ_POSITIONS or {}
|
||||||
@@ -576,13 +580,31 @@ local function evaluate_gpr_value_rule(rule, ev_args, gpr_values)
|
|||||||
return shift_left_u4(immediate % 0x10000, 16)
|
return shift_left_u4(immediate % 0x10000, 16)
|
||||||
end
|
end
|
||||||
|
|
||||||
local source = nil
|
-- Encoders that take `R_0` implicitly (e.g. `li_s(rt, imm)` which is `add_ui(rt, R_0, imm)`) have a non-GPR operand at the source position.
|
||||||
|
-- Fall back to R_0 = 0.
|
||||||
|
-- The implicit-R_0 macros also use a different immediate position (e.g. `li_s`'s `add_ui` rule has source = 2 / immediate = 3
|
||||||
|
-- but the macro takes 2 args); when the configured immediate position is out of bounds.
|
||||||
|
-- Fall back instead to scanning the macro's args for the first integer literal and use that as the immediate.
|
||||||
|
local source = 0
|
||||||
if rule.source then
|
if rule.source then
|
||||||
source = constant_for_operand(gpr_values, ev_args[rule.source])
|
if is_gpr_operand(ev_args[rule.source]) then
|
||||||
if source == nil then return nil end
|
source = constant_for_operand(gpr_values, ev_args[rule.source])
|
||||||
|
if source == nil then return nil end
|
||||||
|
end
|
||||||
|
-- Non-GPR at source position = implicit R_0; source stays 0.
|
||||||
|
end
|
||||||
|
local immediate = nil
|
||||||
|
if rule.immediate and ev_args[rule.immediate] ~= nil then
|
||||||
|
immediate = parse_integer_literal(ev_args[rule.immediate])
|
||||||
|
if immediate == nil then return nil end
|
||||||
|
elseif rule.immediate then
|
||||||
|
-- Immediate position out of bounds: scan for the first integer literal in the args.
|
||||||
|
for _, arg in ipairs(ev_args) do
|
||||||
|
immediate = parse_integer_literal(arg)
|
||||||
|
if immediate ~= nil then break end
|
||||||
|
end
|
||||||
|
if immediate == nil then return nil end
|
||||||
end
|
end
|
||||||
local immediate = rule.immediate and parse_integer_literal(ev_args[rule.immediate]) or nil
|
|
||||||
if rule.immediate and immediate == nil then return nil end
|
|
||||||
if operation == "add_ui" then return wrap_u4( source + sign_extend_i16(immediate))
|
if operation == "add_ui" then return wrap_u4( source + sign_extend_i16(immediate))
|
||||||
elseif operation == "or_i" then return bit_binary( source, immediate % 0x10000, "or")
|
elseif operation == "or_i" then return bit_binary( source, immediate % 0x10000, "or")
|
||||||
elseif operation == "and_i" then return bit_binary( source, immediate % 0x10000, "and")
|
elseif operation == "and_i" then return bit_binary( source, immediate % 0x10000, "and")
|
||||||
@@ -718,8 +740,8 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
|
|||||||
local transition = forward.cu2_transition
|
local transition = forward.cu2_transition
|
||||||
if not transition then return end
|
if not transition then return end
|
||||||
|
|
||||||
local gap = ev_word - transition.producer_word - 1
|
local gap = ev_word - transition.producer_word - 1
|
||||||
local target = transition.target_state
|
local target = transition.target_state
|
||||||
local event_line = line_for_word_event(event)
|
local event_line = line_for_word_event(event)
|
||||||
if target == "unknown" then
|
if target == "unknown" then
|
||||||
append_cu2_finding(atom, event, forward, transition, gap, "info", "unknown",
|
append_cu2_finding(atom, event, forward, transition, gap, "info", "unknown",
|
||||||
@@ -752,8 +774,7 @@ local function consume_cu2_transition(atom, event, ev_word, forward)
|
|||||||
else
|
else
|
||||||
append_cu2_finding(atom, event, forward, transition, gap,
|
append_cu2_finding(atom, event, forward, transition, gap,
|
||||||
"error", "exact",
|
"error", "exact",
|
||||||
string.format(
|
string.format("%s at line %d: COP2 unavailable after SR.CU2 was disabled"
|
||||||
"%s at line %d: COP2 unavailable after SR.CU2 was disabled"
|
|
||||||
.. " (gap=%d, required=%d)",
|
.. " (gap=%d, required=%d)",
|
||||||
atom.name, event_line,
|
atom.name, event_line,
|
||||||
gap, transition.required))
|
gap, transition.required))
|
||||||
@@ -831,8 +852,8 @@ local function analyze_hardware_relations(atom)
|
|||||||
local is_match = false
|
local is_match = false
|
||||||
if semantic == "MTC2" or semantic == "CTC2" or semantic == "LWC2_to_GTE" or semantic == "LWC2_to_other" then
|
if semantic == "MTC2" or semantic == "CTC2" or semantic == "LWC2_to_GTE" or semantic == "LWC2_to_other" then
|
||||||
-- Consumer is a GTE command whose input set contains the producer's COP2 destination (or a fan-out target).
|
-- Consumer is a GTE command whose input set contains the producer's COP2 destination (or a fan-out target).
|
||||||
-- LWC2_to_GTE — GTE-command consumer: gap = 0 OK (the pipeline latches the LWC2 result).
|
-- LWC2_to_GTE — GTE-command consumer: gap = 0 OK (the pipeline latches the LWC2 result).
|
||||||
-- LWC2_to_other — non-GTE consumer: standard load delay applies.
|
-- LWC2_to_other — non-GTE consumer: standard load delay applies.
|
||||||
if relation.id == "lwc2_to_gte_command" then
|
if relation.id == "lwc2_to_gte_command" then
|
||||||
is_match = is_gte_command(ev) and is_cop2_consumer_of(ev, prod.destination, relation)
|
is_match = is_gte_command(ev) and is_cop2_consumer_of(ev, prod.destination, relation)
|
||||||
elseif relation.id == "lwc2_to_other_consumer" then
|
elseif relation.id == "lwc2_to_other_consumer" then
|
||||||
@@ -1266,21 +1287,26 @@ local function check_hazard_nop_use(atom, _pipe_ctx, findings)
|
|||||||
else
|
else
|
||||||
-- Track the slot_kind so the BD-separation case can assert the mac_yield handshake is still suppressed.
|
-- Track the slot_kind so the BD-separation case can assert the mac_yield handshake is still suppressed.
|
||||||
local slot_kind = "plain"
|
local slot_kind = "plain"
|
||||||
-- MIPS load-delay slot: a `load_*` wrote a register in the previous slot, and the result
|
-- MIPS load-delay slot: a `load_*` wrote a register in the previous slot, and the result is unavailable for 1 cycle.
|
||||||
-- is unavailable for 1 cycle. This `nop` is structurally required; classifying it as
|
-- This `nop` is structurally required; classifying it as `modeled-required` is the correct signal
|
||||||
-- `modeled-required` is the correct signal (removing it would make the following
|
-- (removing it would make the following instruction read the OLD value of the loaded register, a load-use hazard).
|
||||||
-- instruction read the OLD value of the loaded register, a load-use hazard). The
|
-- The `load_delay_violations` check (Concern 3) catches the actual read-side error; here we suppress the `modeled-redundant` misclassification.
|
||||||
-- `load_delay_violations` check (Concern 3) catches the actual read-side error; here
|
|
||||||
-- we suppress the `modeled-redundant` misclassification.
|
|
||||||
-- The set of load instructions mirrors the LOAD_INSTRUCTION_IDENTS in `check_load_delay_slots`.
|
-- The set of load instructions mirrors the LOAD_INSTRUCTION_IDENTS in `check_load_delay_slots`.
|
||||||
local load_idents = { load_word = true, load_half = true, load_half_u = true,
|
local load_idents = {
|
||||||
load_byte = true, load_byte_u = true, gte_lw = true, gte_lwc2 = true }
|
load_word = true,
|
||||||
|
load_half = true,
|
||||||
|
load_half_u = true,
|
||||||
|
load_byte = true,
|
||||||
|
load_byte_u = true,
|
||||||
|
gte_lw = true,
|
||||||
|
gte_lwc2 = true
|
||||||
|
}
|
||||||
local is_load_delay = load_idents[prev_ident] == true
|
local is_load_delay = load_idents[prev_ident] == true
|
||||||
if is_load_delay then
|
if is_load_delay then
|
||||||
-- Determine the destination register from the load's `writes` field.
|
-- Determine the destination register from the load's `writes` field.
|
||||||
local prev_writes = gpr_effects[prev_ident] and gpr_effects[prev_ident].writes or {}
|
local prev_writes = gpr_effects[prev_ident] and gpr_effects[prev_ident].writes or {}
|
||||||
local prev_args = prev_ev.args or {}
|
local prev_args = prev_ev.args or {}
|
||||||
local load_dest = prev_writes[1] and prev_args[prev_writes[1]] or "<load-destination>"
|
local load_dest = prev_writes[1] and prev_args[prev_writes[1]] or "<load-destination>"
|
||||||
findings[#findings + 1] = {
|
findings[#findings + 1] = {
|
||||||
check = "hazard_nop_use",
|
check = "hazard_nop_use",
|
||||||
kind = "info",
|
kind = "info",
|
||||||
@@ -1438,17 +1464,21 @@ end
|
|||||||
--- The register becomes non-volatile again at word N+2 (the load has retired), OR sooner if a non-load instruction overwrites the register
|
--- The register becomes non-volatile again at word N+2 (the load has retired), OR sooner if a non-load instruction overwrites the register
|
||||||
--- (the overwriter's write is the fresh producer; the load's value is shadowed and never observed by any reader).
|
--- (the overwriter's write is the fresh producer; the load's value is shadowed and never observed by any reader).
|
||||||
---
|
---
|
||||||
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt: their internal load-then-use sequences
|
--- Runtime-helper atoms / components (`debug_skip == true`) are exempt from some checks, but load-delay
|
||||||
--- are part of the fixed handshake (e.g. `ac_load_tri_indices` loads into R_T0..R_T2, but those are caller-supplied).
|
--- safety applies to their emitted instructions as well.
|
||||||
---
|
---
|
||||||
--- The walker reads `duffle.OPERAND_READ_POSITIONS[event.encoder]` to determine which args are read-source
|
--- The walker reads `duffle.OPERAND_READ_POSITIONS[event.encoder]` to determine which args are read-source
|
||||||
--- (the destination of a load is in `writes`, not `reads` — see `duffle.INSTRUCTION_GPR_EFFECTS`).
|
--- (the destination of a load is in `writes`, not `reads` — see `duffle.INSTRUCTION_GPR_EFFECTS`).
|
||||||
--- The check is purely structural; it does not consult the GPR-value lattice (no constant propagation needed for load-delay detection — the volatility window is unconditional).
|
--- The check is purely structural; it does not consult the GPR-value lattice
|
||||||
|
--- (no constant propagation needed for load-delay detection — the volatility window is unconditional).
|
||||||
local function check_load_delay_slots(atom, pipe_ctx, findings)
|
local function check_load_delay_slots(atom, pipe_ctx, findings)
|
||||||
if atom.kind ~= "atom" then return end
|
-- The load-delay check applies to every atom and component body, including debug-skipped components (`ac_*` and `atom_dbg_skip MipsAtom_(...)`).
|
||||||
local events = atom.paths.word_events or {}
|
-- The `atom_dbg_skip` marker controls debugger stepping, not instruction safety.
|
||||||
if #events == 0 then return end
|
-- `atom_proc` atoms have full bodies with loads that need delay slots, so the check applies to them too.
|
||||||
if is_runtime_helper(atom) then return end
|
local p = atom.paths or {}
|
||||||
|
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||||
|
local events = p.word_events or {}
|
||||||
|
if #events == 0 then return end
|
||||||
|
|
||||||
local gpr_effects = duffle.INSTRUCTION_GPR_EFFECTS or {}
|
local gpr_effects = duffle.INSTRUCTION_GPR_EFFECTS or {}
|
||||||
local read_positions = duffle.OPERAND_READ_POSITIONS or {}
|
local read_positions = duffle.OPERAND_READ_POSITIONS or {}
|
||||||
@@ -1550,6 +1580,8 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
|||||||
if is_runtime_helper(atom) then return end
|
if is_runtime_helper(atom) then return end
|
||||||
-- Per-kind semantics:
|
-- Per-kind semantics:
|
||||||
-- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job.
|
-- MipsAtom_ (baked atom): exactly 1 mac_yield at the end of the body. Control transfer is the atom's job.
|
||||||
|
-- MipsAtom_Proc_ (runtime-proc atom): exactly 1 mac_yield at the end of the body. Same as baked atom;
|
||||||
|
-- the proc IS the atom; the runtime call to `atombuilder_unroll` doesn't introduce a parent atom.
|
||||||
-- MipsAtomComp_ (bare static-array component): ZERO mac_yield.
|
-- MipsAtomComp_ (bare static-array component): ZERO mac_yield.
|
||||||
-- The component is invoked from inside an atom body; the parent atom does the yield.
|
-- The component is invoked from inside an atom body; the parent atom does the yield.
|
||||||
-- MipsAtomComp_Proc_ (procedural component): ZERO mac_yield.
|
-- MipsAtomComp_Proc_ (procedural component): ZERO mac_yield.
|
||||||
@@ -1573,7 +1605,7 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
|||||||
return atom.line + line_in_body[tokens[idx].rel]
|
return atom.line + line_in_body[tokens[idx].rel]
|
||||||
end
|
end
|
||||||
|
|
||||||
if atom.kind == "atom" then
|
if atom.kind == "atom" or atom.kind == "atom_proc" then
|
||||||
-- Baked atom: exactly 1 yield at the end.
|
-- Baked atom: exactly 1 yield at the end.
|
||||||
if count == 0 then
|
if count == 0 then
|
||||||
findings[#findings + 1] = {
|
findings[#findings + 1] = {
|
||||||
@@ -1618,6 +1650,7 @@ local function check_mac_yield_uniformity(atom, pipe_ctx, findings)
|
|||||||
-- The parent atom does the yield.
|
-- The parent atom does the yield.
|
||||||
-- A yield inside a component would either be dead code (bare) or prematurely terminate the function (proc).
|
-- A yield inside a component would either be dead code (bare) or prematurely terminate the function (proc).
|
||||||
-- Both are bugs.
|
-- Both are bugs.
|
||||||
|
-- `atom_proc` atoms are NOT components; they're runtime-proc atoms that own their own yield (handled in the `if` branch above).
|
||||||
if count > 0 then
|
if count > 0 then
|
||||||
findings[#findings + 1] = {
|
findings[#findings + 1] = {
|
||||||
atom = atom.name,
|
atom = atom.name,
|
||||||
@@ -1649,7 +1682,7 @@ end
|
|||||||
--- Per-atom. Runtime-helper atoms (`debug_skip`) are exempt.
|
--- Per-atom. Runtime-helper atoms (`debug_skip`) are exempt.
|
||||||
--- Takes `(atom, pipe_ctx, findings)`; `pipe_ctx` is unused.
|
--- Takes `(atom, pipe_ctx, findings)`; `pipe_ctx` is unused.
|
||||||
local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
||||||
if atom.kind ~= "atom" then return end
|
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||||
if is_runtime_helper(atom) then return end
|
if is_runtime_helper(atom) then return end
|
||||||
|
|
||||||
local tokens = atom.paths.tokens
|
local tokens = atom.paths.tokens
|
||||||
@@ -1661,21 +1694,39 @@ local function check_yield_load_tail_pairing(atom, _pipe_ctx, findings)
|
|||||||
return atom.line + line_in_body[tokens[idx].rel]
|
return atom.line + line_in_body[tokens[idx].rel]
|
||||||
end
|
end
|
||||||
|
|
||||||
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot.
|
-- ── Rule 1: every `mac_yield_load()` must be in a branch BD-slot, OR sit between two `atom_label`s (natural fall-through load pattern).
|
||||||
|
-- When the pattern is satisfied, the check stays silent; only violations emit findings.
|
||||||
for tok_idx = 1, n do
|
for tok_idx = 1, n do
|
||||||
local c = tc[tok_idx]
|
local c = tc[tok_idx]
|
||||||
if c.ident == "mac_yield_load" then
|
if c.ident == "mac_yield_load" then
|
||||||
if tok_idx < 2 or not tc[tok_idx - 1].is_branch then
|
local prev_tc = (tok_idx >= 2) and tc[tok_idx - 1] or nil
|
||||||
local prev_ident = (tok_idx >= 2) and (tc[tok_idx - 1].ident or "?") or "<none>"
|
-- Look for the next `atom_label()` token (skip `atom_offset` markers; check immediately-adjacent first).
|
||||||
findings[#findings + 1] = {
|
local next_label_tc = (tok_idx + 1 <= n) and tc[tok_idx + 1] or nil
|
||||||
atom = atom.name,
|
if next_label_tc and next_label_tc.ident ~= "atom_label" then
|
||||||
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
|
next_label_tc = nil
|
||||||
check = "yield_load_tail_pairing",
|
for j = tok_idx + 1, n do
|
||||||
kind = "error",
|
local t = tc[j]
|
||||||
msg = string.format(
|
if t.ident == "atom_label" then
|
||||||
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — `mac_yield_load()` must fill a branch BD-slot."
|
next_label_tc = t
|
||||||
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident),
|
break
|
||||||
}
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
local natural_fallthrough = prev_tc and prev_tc.is_atom_label and next_label_tc ~= nil
|
||||||
|
if not natural_fallthrough then
|
||||||
|
if tok_idx < 2 or not prev_tc.is_branch then
|
||||||
|
local prev_ident = prev_tc and (prev_tc.ident or "?") or "<none>"
|
||||||
|
local next_ident = next_label_tc and (next_label_tc.ident .. "(" .. (next_label_tc.label_name or "?") .. ")") or "<no following label>"
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name,
|
||||||
|
line = tok_idx >= 2 and line_for(tok_idx) or atom.line,
|
||||||
|
check = "yield_load_tail_pairing",
|
||||||
|
kind = "error",
|
||||||
|
msg = string.format(
|
||||||
|
"%s at line %d has `mac_yield_load()` at word %d but the previous token is `%s`, not a branch — and the next `atom_label()` token is `%s` — `mac_yield_load()` must fill a branch BD-slot or sit between two `atom_label`s for the natural fall-through load."
|
||||||
|
, atom.name, tok_idx >= 2 and line_for(tok_idx) or atom.line, tok_idx, prev_ident, next_ident),
|
||||||
|
}
|
||||||
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -1850,9 +1901,9 @@ end
|
|||||||
--- - Atoms containing a `mac_<name>(...)` call whose `name` is not registered in `pipe_ctx.components_by_name` emit a "new macro;
|
--- - Atoms containing a `mac_<name>(...)` call whose `name` is not registered in `pipe_ctx.components_by_name` emit a "new macro;
|
||||||
--- Not in corpus.components" advisory — the auto-derivation returned nil for that name.
|
--- Not in corpus.components" advisory — the auto-derivation returned nil for that name.
|
||||||
---
|
---
|
||||||
--- Applies only to `kind = "atom"` (baked atoms). Components don't emit full primitives.
|
--- Applies only to `kind = "atom"` or `kind = "atom_proc"` (full-atom bodies). Components don't emit full primitives.
|
||||||
local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
|
local function check_gpu_portstore_shape(atom, pipe_ctx, findings)
|
||||||
if atom.kind ~= "atom" then return end
|
if atom.kind ~= "atom" and atom.kind ~= "atom_proc" then return end
|
||||||
local tokens = atom.paths.tokens
|
local tokens = atom.paths.tokens
|
||||||
local line_in_body = atom.paths.line_in_body
|
local line_in_body = atom.paths.line_in_body
|
||||||
local tc = atom.paths.tok_class
|
local tc = atom.paths.tok_class
|
||||||
@@ -2020,8 +2071,9 @@ local function analyze_atom_paths(atom, pipe_ctx)
|
|||||||
succ[#succ + 1] = label_pos + 1
|
succ[#succ + 1] = label_pos + 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
-- For literal-offset jumps (label == false), the target is a non-tracked address; conservatively omit.
|
-- For literal-offset jumps (label == false), control transfers out unconditionally.
|
||||||
return succ, nil
|
-- Treat as a terminator so the path is recorded (NOT as a silent fall-through to the next token, which is unreachable in this atom's execution).
|
||||||
|
return {}, tok_idx
|
||||||
end
|
end
|
||||||
-- Conditional branch: BD slot absorbed; two successors — fall-through (tok_idx+2) + taken (if known).
|
-- Conditional branch: BD slot absorbed; two successors — fall-through (tok_idx+2) + taken (if known).
|
||||||
if tok_idx + 2 <= n then
|
if tok_idx + 2 <= n then
|
||||||
@@ -2037,9 +2089,11 @@ local function analyze_atom_paths(atom, pipe_ctx)
|
|||||||
-- Return (succ, nil), the second value is the terminator marker (nil = not a terminator).
|
-- Return (succ, nil), the second value is the terminator marker (nil = not a terminator).
|
||||||
return succ, nil
|
return succ, nil
|
||||||
end
|
end
|
||||||
-- Normal token: just the next one
|
-- Normal token: just the next one.
|
||||||
|
-- The final ordinary word of the body has no successor and terminates the path;
|
||||||
|
-- record it as an implicit endpoint so the cycle budget for non-yield components is not silently zeroed.
|
||||||
if tok_idx + 1 <= n then return { tok_idx + 1 }, nil end
|
if tok_idx + 1 <= n then return { tok_idx + 1 }, nil end
|
||||||
return {}, nil
|
return {}, tok_idx
|
||||||
end
|
end
|
||||||
|
|
||||||
-- DFS through all paths. Track the current cycle sum, a visited set scoped to the current path (to detect loops), and a count of paths.
|
-- DFS through all paths. Track the current cycle sum, a visited set scoped to the current path (to detect loops), and a count of paths.
|
||||||
@@ -2334,6 +2388,274 @@ end
|
|||||||
|
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
-- GTE control-register alias + RT-diagonal + TR-naming helpers and checks
|
||||||
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
|
--- Resolve a `gte_cr_<Alias>` ident to its alias-group entry, or nil if the alias
|
||||||
|
--- is in a distinct-slot group (or the alias name is not a known C2 control-register alias).
|
||||||
|
--- Reads `M.GTE_CR_ALIAS_GROUPS` from `duffle.lua`.
|
||||||
|
local function find_alias_pair_for(alias_name, duffle)
|
||||||
|
local groups = (duffle and duffle.GTE_CR_ALIAS_GROUPS) or {}
|
||||||
|
for _, group in ipairs(groups) do
|
||||||
|
for _, name in ipairs(group[2] or {}) do
|
||||||
|
if name == alias_name then return group end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil
|
||||||
|
end
|
||||||
|
|
||||||
|
-- True iff `c` (a TokClass entry) is a CPU→COP2 control-register transfer
|
||||||
|
-- (`gte_mv_to_ctrl_r` / `gte_mv_from_ctrl_r`).
|
||||||
|
local function is_ctrl_r_transfer(c)
|
||||||
|
if c == nil then return false end
|
||||||
|
return c.ident == "gte_mv_to_ctrl_r" or c.ident == "gte_mv_from_ctrl_r"
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Resolve a token's source line. The per-token `line` is the body-relative
|
||||||
|
-- line; `atom.line` is the source line of the atom declaration; `line_in_body`
|
||||||
|
-- (atom.paths) maps a body-relative line to its source line. The arithmetic
|
||||||
|
-- `atom.line + line_in_body[tok.rel] - 1` matches the convention used by
|
||||||
|
-- check_abi_handoff and check_control_transfer_delay_slot_use elsewhere.
|
||||||
|
local function atom_body_token_source_line(atom, token, line_in_body)
|
||||||
|
if line_in_body == nil or token == nil or token.rel == nil then
|
||||||
|
return atom.line or 0
|
||||||
|
end
|
||||||
|
local body_line = line_in_body[token.rel]
|
||||||
|
if body_line == nil then return atom.line or 0 end
|
||||||
|
return (atom.line or 0) + body_line - 1
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check #N: gte_cr_alias_writes
|
||||||
|
-- Fires one warning per atom per alias-group when the atom body touches two
|
||||||
|
-- distinct aliases from the same group. Aliases within a group write to the
|
||||||
|
-- same C2 control-register slot on real silicon; cross-alias writes inside
|
||||||
|
-- one atom body silently clobber each other.
|
||||||
|
--
|
||||||
|
-- Severity: warning. Build continues. The libgte outer-product convention
|
||||||
|
-- uses only RT-row aliases (which are NOT in `M.GTE_CR_ALIAS_GROUPS`), so
|
||||||
|
-- the canonical convention does not trigger this check.
|
||||||
|
local function check_gte_cr_alias_writes(atom, pipe_ctx, findings)
|
||||||
|
local groups = pipe_ctx.gte_cr_alias_groups or {}
|
||||||
|
if not next(groups) then return end
|
||||||
|
|
||||||
|
local tokens = atom.paths and atom.paths.tokens or {}
|
||||||
|
local tc = atom.paths and atom.paths.tok_class or {}
|
||||||
|
local line_in_body = atom.paths and atom.paths.line_in_body
|
||||||
|
if not next(tokens) then return end
|
||||||
|
|
||||||
|
-- Build a per-group set of (alias, source_line) pairs touched in this atom body.
|
||||||
|
-- Walks every token; when the token is a ctrl-r transfer, the alias is at
|
||||||
|
-- position tok_idx + 2 (rt, alias, [imm-or-arg]). The pre-classified
|
||||||
|
-- `tc` table tells us whether the token is a ctrl-r transfer and what its
|
||||||
|
-- source line is.
|
||||||
|
local touched = {}
|
||||||
|
for tok_idx, token in ipairs(tokens) do
|
||||||
|
local c = tc[tok_idx]
|
||||||
|
if is_ctrl_r_transfer(c) and tokens[tok_idx + 2] then
|
||||||
|
local alias = tokens[tok_idx + 2].tok
|
||||||
|
local group = find_alias_pair_for(alias, pipe_ctx.duffle)
|
||||||
|
if group then
|
||||||
|
touched[group[1]] = touched[group[1]] or {}
|
||||||
|
touched[group[1]][#touched[group[1]] + 1] = {
|
||||||
|
alias = alias,
|
||||||
|
line = atom_body_token_source_line(atom, token, line_in_body),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Fire one warning per group touched with 2+ distinct aliases.
|
||||||
|
for slot, hits in pairs(touched) do
|
||||||
|
local seen = {}
|
||||||
|
local distinct = {}
|
||||||
|
for _, h in ipairs(hits) do
|
||||||
|
if not seen[h.alias] then
|
||||||
|
seen[h.alias] = true
|
||||||
|
distinct[#distinct + 1] = h
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if #distinct >= 2 then
|
||||||
|
local aliases = {}
|
||||||
|
for _, d in ipairs(distinct) do aliases[#aliases + 1] = d.alias end
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name or "",
|
||||||
|
line = distinct[1].line,
|
||||||
|
check = "gte_cr_alias_writes",
|
||||||
|
kind = "warning",
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' touches %d aliases that share C2[%d]: %s; verify the intent"
|
||||||
|
, atom.name or "", #distinct, slot, table.concat(aliases, ", ")),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check #N+1: rtdiagonal_completeness
|
||||||
|
-- Fires one info per atom body when the bare `gte_cmdw_mvmva` macro is used.
|
||||||
|
-- The bare macro encodes only the cmd field; the canonical libgte-2-pass
|
||||||
|
-- shape uses `gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012` (gte.h:430).
|
||||||
|
--
|
||||||
|
-- Severity: info by default. Escalates to warning when
|
||||||
|
-- `GTE_RT_DIAGONAL_STRICT=1` env var is set (CI / production builds).
|
||||||
|
--
|
||||||
|
-- The bare macro IS the right call for the canonical libgte outer-product
|
||||||
|
-- convention, so this is an opt-out hint rather than a hard warning.
|
||||||
|
local function check_rtdiagonal_completeness(atom, _pipe_ctx, findings)
|
||||||
|
local tokens = atom.paths and atom.paths.tokens or {}
|
||||||
|
local tc = atom.paths and atom.paths.tok_class or {}
|
||||||
|
local line_in_body = atom.paths and atom.paths.line_in_body
|
||||||
|
if not next(tokens) then return end
|
||||||
|
local strict = os.getenv("GTE_RT_DIAGONAL_STRICT") == "1"
|
||||||
|
for tok_idx, token in ipairs(tokens) do
|
||||||
|
local c = tc[tok_idx]
|
||||||
|
if c and c.ident == "gte_cmdw_mvmva" then
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name or "",
|
||||||
|
line = atom_body_token_source_line(atom, token, line_in_body),
|
||||||
|
check = "rtdiagonal_completeness",
|
||||||
|
kind = strict and "warning" or "info",
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' uses the bare gte_cmdw_mvmva macro; "
|
||||||
|
.. "the canonical libgte-2-pass shape is gte_cmdw_mvmva_c11_pass2_exact = 0x4A49E012 "
|
||||||
|
.. "(gte.h:430). The bare macro does not encode RT23/RT31/RT32/RT33; "
|
||||||
|
.. "for a full 3x3 matrix, use the dedicated literal or hand-build via enc_gte_*()."
|
||||||
|
, atom.name or ""),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- Check #N+2: gte_cr_TR_naming
|
||||||
|
-- Fires one info per atom body when a `gte_cr_TR[XYZ]` alias is used.
|
||||||
|
-- Translation-vector registers are the only 3-letter-suffix C2 aliases
|
||||||
|
-- (`TRX/TRY/TRZ`); an agent who reads `TRX` might typo it as `RT_X` or
|
||||||
|
-- `RTX0` and either get a compile error (best case) or a build that
|
||||||
|
-- links but routes the `ctc2` write to the wrong C2 slot.
|
||||||
|
--
|
||||||
|
-- Severity: info. The convention is correct; this is a documentation-pointer check.
|
||||||
|
local function check_gte_cr_TR_naming(atom, _pipe_ctx, findings)
|
||||||
|
local tokens = atom.paths and atom.paths.tokens or {}
|
||||||
|
local tc = atom.paths and atom.paths.tok_class or {}
|
||||||
|
local line_in_body = atom.paths and atom.paths.line_in_body
|
||||||
|
if not next(tokens) then return end
|
||||||
|
local touched = false
|
||||||
|
local first_line = 0
|
||||||
|
for tok_idx, token in ipairs(tokens) do
|
||||||
|
local c = tc[tok_idx]
|
||||||
|
if c and c.ident and c.ident:match("^gte_cr_TR[XYZ]$") then
|
||||||
|
touched = true
|
||||||
|
if first_line == 0 then
|
||||||
|
first_line = atom_body_token_source_line(atom, token, line_in_body)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
if touched then
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
atom = atom.name or "",
|
||||||
|
line = first_line,
|
||||||
|
check = "gte_cr_TR_naming",
|
||||||
|
kind = "info",
|
||||||
|
msg = string.format(
|
||||||
|
"atom '%s' uses gte_cr_TR[XYZ]; translation-vector registers are the only "
|
||||||
|
.. "3-letter-suffix C2 aliases (TRX/TRY/TRZ). See docs/gte_reference.md §"
|
||||||
|
.. "\"The `gte_cmdw_mvmva_c11_pass2_exact` literal\" for the libgte outer-product "
|
||||||
|
.. "convention that uses these names."
|
||||||
|
, atom.name or ""),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- check_immediate_field_width — flags integer literals passed to instruction
|
||||||
|
-- macros that exceed the immediate field width. Reads `IMMEDIATE_FIELD_WIDTHS`
|
||||||
|
-- from duffle.lua. Only fires on parseable integer literals; register names,
|
||||||
|
-- O_(...) offsets, atom_offset(...) markers, and enum tokens are skipped.
|
||||||
|
local function check_immediate_field_width(atom, pipe_ctx, findings)
|
||||||
|
local widths = duffle.IMMEDIATE_FIELD_WIDTHS or {}
|
||||||
|
local events = atom.paths and atom.paths.word_events or {}
|
||||||
|
local line_for_word_event = pipe_ctx.line_for_word_event
|
||||||
|
for _, ev in ipairs(events) do
|
||||||
|
local ev_ident = ev.encoder or ev.ident or "?"
|
||||||
|
local rules = widths[ev_ident]
|
||||||
|
if rules then
|
||||||
|
local ev_args = ev.args or {}
|
||||||
|
local ev_line = line_for_word_event and line_for_word_event(ev) or atom.line
|
||||||
|
for _, rule in ipairs(rules) do
|
||||||
|
local arg_str = ev_args[rule.arg]
|
||||||
|
if arg_str then
|
||||||
|
local value = parse_integer_literal(arg_str)
|
||||||
|
if value then
|
||||||
|
local width = rule.width
|
||||||
|
local is_signed = rule.signed == true
|
||||||
|
-- parse_integer_literal returns a U4-wrapped value in [0, 2^32).
|
||||||
|
-- For signed fields, re-interpret the high bit as the sign.
|
||||||
|
local signed_value = value
|
||||||
|
if is_signed and value >= 0x80000000 then
|
||||||
|
signed_value = value - 0x100000000
|
||||||
|
end
|
||||||
|
local lo, hi
|
||||||
|
if is_signed then
|
||||||
|
lo = -(bit.lshift(1, width - 1))
|
||||||
|
hi = bit.lshift(1, width - 1) - 1
|
||||||
|
else
|
||||||
|
lo = 0
|
||||||
|
hi = bit.lshift(1, width) - 1
|
||||||
|
end
|
||||||
|
-- For unsigned fields, a negative C literal (high bit set in U4)
|
||||||
|
-- is valid if the low `width` bits fit — IMM_MASK truncates it.
|
||||||
|
-- Flag as a warning (code smell), not an error.
|
||||||
|
local check_value = is_signed and signed_value or value
|
||||||
|
local field_max = bit.lshift(1, width) - 1
|
||||||
|
local low_bits_fit = (value % (bit.lshift(1, width))) == value or (is_signed and signed_value >= lo and signed_value <= hi)
|
||||||
|
if is_signed then
|
||||||
|
if signed_value < lo or signed_value > hi then
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
check = "immediate_field_width",
|
||||||
|
kind = "error",
|
||||||
|
atom = atom.name,
|
||||||
|
line = ev_line,
|
||||||
|
msg = string.format(
|
||||||
|
"%s: immediate %d at arg %d overflows %d-bit %s field (valid %d..%d)",
|
||||||
|
ev_ident, signed_value, rule.arg, width,
|
||||||
|
"signed", lo, hi),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
else
|
||||||
|
-- Unsigned field: check if the low `width` bits exceed the field.
|
||||||
|
-- A negative C literal (U4 >= 0x80000000) whose low bits fit is
|
||||||
|
-- valid but a code smell — warn, don't error.
|
||||||
|
local low_bits = value % (bit.lshift(1, width))
|
||||||
|
if value > field_max then
|
||||||
|
if value >= 0x80000000 and low_bits <= field_max then
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
check = "immediate_field_width",
|
||||||
|
kind = "warning",
|
||||||
|
atom = atom.name,
|
||||||
|
line = ev_line,
|
||||||
|
msg = string.format(
|
||||||
|
"%s: negative immediate %d at arg %d on unsigned %d-bit field (truncated to %d by IMM_MASK)",
|
||||||
|
ev_ident, signed_value, rule.arg, width, low_bits),
|
||||||
|
}
|
||||||
|
else
|
||||||
|
findings[#findings + 1] = {
|
||||||
|
check = "immediate_field_width",
|
||||||
|
kind = "error",
|
||||||
|
atom = atom.name,
|
||||||
|
line = ev_line,
|
||||||
|
msg = string.format(
|
||||||
|
"%s: immediate %d at arg %d overflows %d-bit unsigned field (valid 0..%d)",
|
||||||
|
ev_ident, value, rule.arg, width, field_max),
|
||||||
|
}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
|
-- CHECK_RULES — data-driven check dispatch (Muratori: data over control flow)
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
@@ -2349,20 +2671,24 @@ end
|
|||||||
-- This is the plex pattern: the iteration is in ONE place (validate), the variation is in DATA (this table).
|
-- This is the plex pattern: the iteration is in ONE place (validate), the variation is in DATA (this table).
|
||||||
|
|
||||||
local CHECK_RULES = {
|
local CHECK_RULES = {
|
||||||
{ name = "transfer_hazards", per_atom = check_transfer_hazards },
|
{ name = "transfer_hazards", per_atom = check_transfer_hazards },
|
||||||
{ name = "gte_input_latch", per_atom = check_gte_input_latch },
|
{ name = "gte_input_latch", per_atom = check_gte_input_latch },
|
||||||
{ name = "gte_role_mismatch", per_atom = check_gte_role_mismatch },
|
{ name = "gte_role_mismatch", per_atom = check_gte_role_mismatch },
|
||||||
{ name = "hazard_nop_use", per_atom = check_hazard_nop_use },
|
{ name = "hazard_nop_use", per_atom = check_hazard_nop_use },
|
||||||
{ name = "control_transfer_delay_slot_use",per_atom = check_control_transfer_delay_slot_use},
|
{ name = "control_transfer_delay_slot_use", per_atom = check_control_transfer_delay_slot_use },
|
||||||
{ name = "load_delay_violation", per_atom = check_load_delay_slots },
|
{ name = "load_delay_violation", per_atom = check_load_delay_slots },
|
||||||
{ name = "mac_yield_uniformity", per_atom = check_mac_yield_uniformity },
|
{ name = "mac_yield_uniformity", per_atom = check_mac_yield_uniformity },
|
||||||
{ name = "yield_load_tail_pairing", per_atom = check_yield_load_tail_pairing },
|
{ name = "yield_load_tail_pairing", per_atom = check_yield_load_tail_pairing },
|
||||||
{ name = "abi_handoff", per_atom = check_abi_handoff },
|
{ name = "abi_handoff", per_atom = check_abi_handoff },
|
||||||
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
|
{ name = "gpu_portstore_shape", per_atom = check_gpu_portstore_shape },
|
||||||
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
|
{ name = "per_atom_cycle_budget", per_atom = check_per_atom_cycle_budget },
|
||||||
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
|
{ name = "gte_cr_alias_writes", per_atom = check_gte_cr_alias_writes },
|
||||||
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
|
{ name = "rtdiagonal_completeness", per_atom = check_rtdiagonal_completeness },
|
||||||
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
|
{ name = "gte_cr_TR_naming", per_atom = check_gte_cr_TR_naming },
|
||||||
|
{ name = "immediate_field_width", per_atom = check_immediate_field_width },
|
||||||
|
{ name = "enum_alias_membership", per_source = check_enum_alias_membership },
|
||||||
|
{ name = "atom_type_consistency", per_source = check_atom_type_consistency },
|
||||||
|
{ name = "binds_no_substruct_deref", per_source = check_binds_no_substruct_deref },
|
||||||
}
|
}
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -2397,12 +2723,17 @@ local function build_corpus_pipe_ctx(ctx)
|
|||||||
atoms_by_name = corpus.atoms_by_name or {},
|
atoms_by_name = corpus.atoms_by_name or {},
|
||||||
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original
|
-- Per-component metadata (cycle_cost + gp0_contrib) auto-derived from the original
|
||||||
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
|
-- `MipsAtomComp_` body by `passes/components.lua::compute_components_metadata`.
|
||||||
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
|
-- Keyed by bare name (e.g. `format_f3_color`, `gte_store_f3`); the `mac_` prefix at call sites is stripped before lookup.
|
||||||
components_by_name = corpus.components or {},
|
components_by_name = corpus.components or {},
|
||||||
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
-- Corpus-wide ordered list of atom_info records (source-order + duplicates).
|
||||||
atom_infos_list = corpus.atom_infos or {},
|
atom_infos_list = corpus.atom_infos or {},
|
||||||
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
-- Corpus-wide collisions (recorded by scan_source.merge_corpus_registries).
|
||||||
collisions = corpus.collisions or {},
|
collisions = corpus.collisions or {},
|
||||||
|
-- GTE control-register alias groups (from `duffle.GTE_CR_ALIAS_GROUPS`).
|
||||||
|
-- The three new per_atom checks (gte_cr_alias_writes, rtdiagonal_completeness,
|
||||||
|
-- gte_cr_TR_naming) read from this view. `duffle` is exposed alongside so
|
||||||
|
-- `find_alias_pair_for` can resolve alias → group without a separate registry.
|
||||||
|
gte_cr_alias_groups = duffle.GTE_CR_ALIAS_GROUPS or {},
|
||||||
}
|
}
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -2521,28 +2852,27 @@ local function validate(ctx, src, corpus_pipe_ctx)
|
|||||||
-- Hazard readers (transfer_hazards) populate `f.check`, `f.relation_id`, `f.semantic`, `f.direction`, `f.producer_destination`, `f.gap`, `f.required`, `f.evidence_confidence`, etc.;
|
-- Hazard readers (transfer_hazards) populate `f.check`, `f.relation_id`, `f.semantic`, `f.direction`, `f.producer_destination`, `f.gap`, `f.required`, `f.evidence_confidence`, etc.;
|
||||||
-- Copying them through keeps the per-severity bucket schema compatible with the renderer while making the diagnostic payload queryable.
|
-- Copying them through keeps the per-severity bucket schema compatible with the renderer while making the diagnostic payload queryable.
|
||||||
local payload = {
|
local payload = {
|
||||||
line = f.line,
|
line = f.line,
|
||||||
msg = f.msg,
|
msg = f.msg,
|
||||||
check = f.check,
|
check = f.check,
|
||||||
atom = f.atom,
|
atom = f.atom,
|
||||||
source = f.source,
|
source = f.source,
|
||||||
relation_id = f.relation_id,
|
relation_id = f.relation_id,
|
||||||
semantic = f.semantic,
|
semantic = f.semantic,
|
||||||
direction = f.direction,
|
direction = f.direction,
|
||||||
producer_destination = f.producer_destination,
|
producer_destination = f.producer_destination,
|
||||||
producer_word = f.producer_word,
|
producer_word = f.producer_word,
|
||||||
producer_line = f.producer_line,
|
producer_line = f.producer_line,
|
||||||
producer_source = f.producer_source,
|
producer_source = f.producer_source,
|
||||||
consumer_word = f.consumer_word,
|
consumer_word = f.consumer_word,
|
||||||
consumer_token = f.consumer_token,
|
consumer_token = f.consumer_token,
|
||||||
gap = f.gap,
|
gap = f.gap,
|
||||||
required = f.required,
|
required = f.required,
|
||||||
evidence_confidence = f.evidence_confidence,
|
evidence_confidence = f.evidence_confidence,
|
||||||
evidence_source = f.evidence_source,
|
evidence_source = f.evidence_source,
|
||||||
}
|
}
|
||||||
-- Preserve relation fields such as target_state and status_register,
|
-- Preserve relation fields such as target_state and status_register status_value,
|
||||||
-- status_value, and future policy metadata) without making the binner
|
-- and future policy metadata) without making the binner another semantic walker.
|
||||||
-- another semantic walker.
|
|
||||||
for key, value in pairs(f) do
|
for key, value in pairs(f) do
|
||||||
if payload[key] == nil then payload[key] = value end
|
if payload[key] == nil then payload[key] = value end
|
||||||
end
|
end
|
||||||
@@ -2624,7 +2954,7 @@ function M.run(ctx)
|
|||||||
-- Build the corpus-wide pipe_ctx ONCE per pass run.
|
-- Build the corpus-wide pipe_ctx ONCE per pass run.
|
||||||
-- The pipe_ctx is shared across every validate() invocation in this M.run so cross-source visibility is constant.
|
-- The pipe_ctx is shared across every validate() invocation in this M.run so cross-source visibility is constant.
|
||||||
local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx)
|
local corpus_pipe_ctx = build_corpus_pipe_ctx(ctx)
|
||||||
local corpus = ctx.shared.corpus
|
local corpus = ctx.shared.corpus
|
||||||
|
|
||||||
-- Aggregate per-DIRECTORY (per-module).
|
-- Aggregate per-DIRECTORY (per-module).
|
||||||
-- One static_analysis.txt per source-directory, emitted only if the directory contains at least one atom.
|
-- One static_analysis.txt per source-directory, emitted only if the directory contains at least one atom.
|
||||||
@@ -2676,8 +3006,8 @@ function M.run(ctx)
|
|||||||
|
|
||||||
-- Aggregate per-dir errors/warnings/info into the orchestrator totals.
|
-- Aggregate per-dir errors/warnings/info into the orchestrator totals.
|
||||||
-- Hoisted out of any per-dir file-emit so `report.lua` can drop the on-disk file emitter without losing the cross-module rollup.
|
-- Hoisted out of any per-dir file-emit so `report.lua` can drop the on-disk file emitter without losing the cross-module rollup.
|
||||||
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end
|
for _, e in ipairs(dir_errors) do errors [#errors + 1] = e end
|
||||||
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end
|
for _, w in ipairs(dir_warnings) do warnings[#warnings + 1] = w end
|
||||||
for _, i_ in ipairs(dir_info) do info [#info + 1] = i_ end
|
for _, i_ in ipairs(dir_info) do info [#info + 1] = i_ end
|
||||||
-- (No per-dir emit: per-module findings are stashed on `corpus.static_analysis_results` above.
|
-- (No per-dir emit: per-module findings are stashed on `corpus.static_analysis_results` above.
|
||||||
-- `report.lua` reads that projection to render `<module>.atom_meta_report.md` without re-running validate().)
|
-- `report.lua` reads that projection to render `<module>.atom_meta_report.md` without re-running validate().)
|
||||||
|
|||||||
@@ -93,7 +93,7 @@ end
|
|||||||
--- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
|
--- Load the authored `word_count.metadata.h` into `ctx.shared.corpus.word_counts`.
|
||||||
--- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
|
--- Generated `.macs.h` files are OUTPUT artifacts and are NOT scanned as inputs.
|
||||||
--- Current component counts are computed and inserted by `passes/components.lua`
|
--- Current component counts are computed and inserted by `passes/components.lua`
|
||||||
--- after the components pass iterates `corpus.source_order` and writes each source's `<dir_basename>.macs.h` file.
|
--- after the components pass iterates `corpus.source_order` and writes each source-directory's `gen/macs.h` file.
|
||||||
---
|
---
|
||||||
--- Contract:
|
--- Contract:
|
||||||
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
|
--- * `ctx.shared.corpus` MUST exist (canonical corpus ownership).
|
||||||
|
|||||||
Binary file not shown.
+45
-50
@@ -16,7 +16,7 @@
|
|||||||
|
|
||||||
-- Bootstrap: load `duffle_paths.lua` via this script's own path.
|
-- Bootstrap: load `duffle_paths.lua` via this script's own path.
|
||||||
-- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua");
|
-- Use `arg[0]` when this file is the entry script (`arg[0]` ends in "ps1_meta.lua");
|
||||||
-- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path, not ours).
|
-- fall back to `debug.getinfo(1, "S").source` when this file is being dofile()'d or require()'d (in which case `arg[0]` is the *caller's* path).
|
||||||
-- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`.
|
-- That single statement: (a) sets `package.path` + `package.cpath`, (b) at the bottom returns `require("duffle")`.
|
||||||
-- So the dofile's return value is the duffle module.
|
-- So the dofile's return value is the duffle module.
|
||||||
local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil
|
local _is_entry_script = arg and arg[0] and arg[0]:match("ps1_meta%.lua$") ~= nil
|
||||||
@@ -54,45 +54,44 @@ local PASS_FLAG_DISPATCH_KEY = "__pass__"
|
|||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
--- @class PassDescriptor
|
--- @class PassDescriptor
|
||||||
--- @field module string -- module name passed to require()
|
--- @field module string -- Module name passed to require()
|
||||||
--- @field kind string -- "shared" | "header-output" | "validation" | "diagnostic" | "report"
|
--- @field kind string -- "shared" | "header-output" | "validation" | "diagnostic" | "report"
|
||||||
--- -- Report severity is independent from process exit policy (see PASS_KIND_STOP_ON_ERROR).
|
--- -- Report severity is independent from process exit policy (see PASS_KIND_STOP_ON_ERROR).
|
||||||
--- @field deps string[] -- names of upstream passes
|
--- @field deps string[] -- Names of upstream passes
|
||||||
--- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of
|
--- @field groups string[]? -- OPTIONAL build-phase groups this pass is a root of (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
|
||||||
--- -- (e.g. { "pre-link" }, { "post-link" }); absent ⇒ dependency-only
|
|
||||||
|
|
||||||
--- @class SourceFile
|
--- @class SourceFile
|
||||||
--- @field path string -- absolute path to the source file
|
--- @field path string -- Absolute path to the source file
|
||||||
--- @field text string -- the full source text
|
--- @field text string -- Full source text
|
||||||
--- @field dir string -- the directory containing the source
|
--- @field dir string -- Directory containing the source
|
||||||
--- @field basename string -- filename without extension
|
--- @field basename string -- Filename without extension
|
||||||
|
|
||||||
--- @class PassCtx
|
--- @class PassCtx
|
||||||
--- @field metadata_path string -- path to word_count.metadata.h
|
--- @field metadata_path string -- Path to word_count.metadata.h
|
||||||
--- @field shared table -- cross-pass shared state
|
--- @field shared table -- Cross-pass shared state
|
||||||
--- @field shared.corpus table -- canonical authored-source/project projection
|
--- @field shared.corpus table -- Authored-source/project projection
|
||||||
--- @field out_root string -- output root (e.g. "build/gen")
|
--- @field out_root string -- Output root (e.g. "build/gen")
|
||||||
--- @field project_root string -- PS1 repository root
|
--- @field project_root string -- PS1 repository root
|
||||||
--- @field flags table -- CLI flags + per-pass stash
|
--- @field flags table -- CLI flags + per-pass stash
|
||||||
--- @field verbose boolean -- if true, log diagnostic info
|
--- @field verbose boolean -- If true, log diagnostic info
|
||||||
|
|
||||||
--- @class Finding
|
--- @class Finding
|
||||||
--- @field line integer -- source line (or 0 for pass-level)
|
--- @field line integer -- Source line (or 0 for pass-level)
|
||||||
--- @field msg string -- finding message
|
--- @field msg string -- Finding message
|
||||||
|
|
||||||
--- @class PassResult
|
--- @class PassResult
|
||||||
--- @field outputs PassOutputEntry[] -- emitted file paths
|
--- @field outputs PassOutputEntry[] -- Emitted file paths
|
||||||
--- @field errors Finding[] -- build-stops (per-pass kind policy)
|
--- @field errors Finding[] -- Build-stops (per-pass kind policy)
|
||||||
--- @field warnings Finding[] -- informational
|
--- @field warnings Finding[] -- Informational
|
||||||
|
|
||||||
--- @class ParsedArgs
|
--- @class ParsedArgs
|
||||||
--- @field requested_set string[] -- pass names to run (explicit --all expanded)
|
--- @field requested_set string[] -- Pass names to run (explicit --all expanded)
|
||||||
--- @field sources string[] -- exact --source values, retained in CLI order
|
--- @field sources string[] -- Exact --source values, retained in CLI order
|
||||||
--- @field unity_root string|nil -- --unity-root value; mutually exclusive with sources
|
--- @field unity_root string|nil -- --unity-root value; mutually exclusive with sources
|
||||||
--- @field metadata string -- --metadata value
|
--- @field metadata string -- --metadata value
|
||||||
--- @field out_root string -- --out-root value (default "build/gen")
|
--- @field out_root string -- --out-root value (default "build/gen")
|
||||||
--- @field project_root string -- PS1 repository root (derived from metadata by default)
|
--- @field project_root string -- PS1 repository root (derived from metadata by default)
|
||||||
--- @field verbose boolean -- if true, log diagnostic info
|
--- @field verbose boolean -- If true, log diagnostic info
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
-- PASSES Table
|
-- PASSES Table
|
||||||
@@ -119,6 +118,12 @@ local PASSES = {
|
|||||||
kind = "header-output",
|
kind = "header-output",
|
||||||
deps = {"scan-source", "word-counts"},
|
deps = {"scan-source", "word-counts"},
|
||||||
},
|
},
|
||||||
|
auto_reg = {
|
||||||
|
module = "passes.auto_reg",
|
||||||
|
kind = "header-output",
|
||||||
|
deps = {"components"},
|
||||||
|
groups = { "pre-link" },
|
||||||
|
},
|
||||||
["emission-model"] = {
|
["emission-model"] = {
|
||||||
module = "passes.emission_model",
|
module = "passes.emission_model",
|
||||||
kind = "validation",
|
kind = "validation",
|
||||||
@@ -138,7 +143,7 @@ local PASSES = {
|
|||||||
["static-analysis"] = {
|
["static-analysis"] = {
|
||||||
module = "passes.static_analysis",
|
module = "passes.static_analysis",
|
||||||
-- "diagnostic" — every `error`/`warning` finding is written to the report file;
|
-- "diagnostic" — every `error`/`warning` finding is written to the report file;
|
||||||
-- the orchestrator does NOT exit non-zero on these findings (see PASS_KIND_STOP_ON_ERROR).
|
-- The orchestrator does NOT exit non-zero on these findings (see PASS_KIND_STOP_ON_ERROR).
|
||||||
-- Report severity is independent from process exit policy.
|
-- Report severity is independent from process exit policy.
|
||||||
kind = "diagnostic",
|
kind = "diagnostic",
|
||||||
deps = {"scan-source", "word-counts", "components", "emission-model"},
|
deps = {"scan-source", "word-counts", "components", "emission-model"},
|
||||||
@@ -163,12 +168,12 @@ local PASSES = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
-- ────────────────────────────────────────────────────────────────────────────
|
-- ────────────────────────────────────────────────────────────────────────────
|
||||||
-- Phase-root selection: derive the sorted set of roots belonging to a named build-phase group, then append them to `args.requested_set`.
|
-- Phase-root selection: Derive the sorted set of roots belonging to a named build-phase group, then append them to `args.requested_set`.
|
||||||
-- topo_sort closes the transitive deps from there; dispatch_passes runs every resolved pass without phase-filtering.
|
-- topo_sort closes the transitive deps from there; dispatch_passes runs every resolved pass without phase-filtering.
|
||||||
-- ────────────────────────────────────────────────────────────────────────────
|
-- ────────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
--- @param group_name string -- the build-phase group ("pre-link" | "post-link")
|
--- @param group_name string -- Build-phase group ("pre-link" | "post-link")
|
||||||
--- @return string[] -- sorted root pass names belonging to that group
|
--- @return string[] -- Sorted root pass names belonging to that group
|
||||||
local function roots_for_group(group_name)
|
local function roots_for_group(group_name)
|
||||||
local names = {}
|
local names = {}
|
||||||
for name, pass in pairs(PASSES) do
|
for name, pass in pairs(PASSES) do
|
||||||
@@ -206,7 +211,7 @@ end
|
|||||||
-- Report severity is independent from process exit policy.
|
-- Report severity is independent from process exit policy.
|
||||||
-- A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
|
-- A "diagnostic" pass still writes every `error`/`warning` finding into its report file,
|
||||||
-- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero.
|
-- but `report_validation_errors` returns early for non-stopping kinds, so nothing is printed to stderr and the orchestrator does not exit non-zero.
|
||||||
-- Adding a new pass kind requires listing it here explicitly; an unknown kind must not silently fall back to "true".
|
-- Adding a new pass kind requires listing it here explicitly; An unknown kind must not silently fall back to "true".
|
||||||
local PASS_KIND_STOP_ON_ERROR = {
|
local PASS_KIND_STOP_ON_ERROR = {
|
||||||
["shared"] = false,
|
["shared"] = false,
|
||||||
["header-output"] = true,
|
["header-output"] = true,
|
||||||
@@ -216,8 +221,7 @@ local PASS_KIND_STOP_ON_ERROR = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
-- Closed set of CLI flags -> pass names.
|
-- Closed set of CLI flags -> pass names.
|
||||||
-- Per-pass flags (e.g. --word-counts) live here; phase flags (--pre-link, --post-link, --all)
|
-- Per-pass flags (e.g. --word-counts); phase flags (--pre-link, --post-link, --all) are within FLAG_HANDLERS because they own side effects or invoke group-derivation logic.
|
||||||
-- live in FLAG_HANDLERS because they own side effects or invoke group-derivation logic.
|
|
||||||
-- dwarf-injection is *also* a per-pass opt-in flag, but its selection + opt-in state are both owned by the explicit FLAG_HANDLERS entry below
|
-- dwarf-injection is *also* a per-pass opt-in flag, but its selection + opt-in state are both owned by the explicit FLAG_HANDLERS entry below
|
||||||
-- (it sets args.flags.dwarf_injection and appends "dwarf-injection" to requested_set), so it is intentionally absent from this table.
|
-- (it sets args.flags.dwarf_injection and appends "dwarf-injection" to requested_set), so it is intentionally absent from this table.
|
||||||
local PASS_FLAG_TO_NAME = {
|
local PASS_FLAG_TO_NAME = {
|
||||||
@@ -246,7 +250,6 @@ end
|
|||||||
|
|
||||||
-- Per-flag handlers. Each handler takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
|
-- Per-flag handlers. Each handler takes (args, argv, arg_idx) and returns the new arg_idx (so multi-arg flags like --source FILE advance it).
|
||||||
-- Returning nil + os.exit() handles termination flags (--help).
|
-- Returning nil + os.exit() handles termination flags (--help).
|
||||||
|
|
||||||
local FLAG_HANDLERS = {}
|
local FLAG_HANDLERS = {}
|
||||||
|
|
||||||
-- ════════════════════════════════════════════════════════════════════════════
|
-- ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -272,9 +275,9 @@ PASS_FLAGS:
|
|||||||
Or pick any subset:
|
Or pick any subset:
|
||||||
--scan-source Scan sources into the fat SourceScan payload
|
--scan-source Scan sources into the fat SourceScan payload
|
||||||
--word-counts Load metadata.h + scan for existing .macs.h
|
--word-counts Load metadata.h + scan for existing .macs.h
|
||||||
--components Generate <module>/gen/<basename>.macs.h
|
--components Generate <srcdir>/gen/macs.h (per-directory aggregation)
|
||||||
--validate Run atom annotation DSL validation
|
--validate Run atom annotation DSL validation
|
||||||
--offsets Generate <module>/gen/<basename>.offsets.h
|
--offsets Generate <srcdir>/gen/offsets.h (per-directory aggregation)
|
||||||
--atoms-source-map Generate <basename>.atoms.sourcemap.txt per source
|
--atoms-source-map Generate <basename>.atoms.sourcemap.txt per source
|
||||||
--dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf.
|
--dwarf-injection [opt-in] Select the post-link dwarf-injection pass + set the opt-in flag. Requires --elf.
|
||||||
--static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget
|
--static-analysis Static analysis: GTE pipeline-fill, mac_yield, ABI handoff, cycle budget
|
||||||
@@ -317,8 +320,7 @@ local function require_flag_value(argv, arg_idx, flag)
|
|||||||
local next_known = type(value) == "string"
|
local next_known = type(value) == "string"
|
||||||
and (FLAG_HANDLERS[value] ~= nil or PASS_FLAG_TO_NAME[value] ~= nil)
|
and (FLAG_HANDLERS[value] ~= nil or PASS_FLAG_TO_NAME[value] ~= nil)
|
||||||
if value == nil or next_known then
|
if value == nil or next_known then
|
||||||
io.stderr:write("ps1_meta: " .. flag .. " requires "
|
io.stderr:write("ps1_meta: " .. flag .. " requires " .. FLAG_VALUE_NAMES[flag] .. "\n")
|
||||||
.. FLAG_VALUE_NAMES[flag] .. "\n")
|
|
||||||
os.exit(EXIT_INTERNAL_ERROR)
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
end
|
end
|
||||||
return value, arg_idx + 1
|
return value, arg_idx + 1
|
||||||
@@ -328,12 +330,10 @@ end
|
|||||||
-- Termination flags like --help call os.exit() instead.
|
-- Termination flags like --help call os.exit() instead.
|
||||||
-- Populated AFTER print_help so the --help handler can reference it as an upvalue (Lua resolves locals at closure-call time,
|
-- Populated AFTER print_help so the --help handler can reference it as an upvalue (Lua resolves locals at closure-call time,
|
||||||
-- but if the closure is defined before the local, it falls back to _G).
|
-- but if the closure is defined before the local, it falls back to _G).
|
||||||
FLAG_HANDLERS["--help"] = function(args)
|
|
||||||
print_help()
|
|
||||||
os.exit(0)
|
|
||||||
end
|
|
||||||
|
|
||||||
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end
|
FLAG_HANDLERS["--help"] = function(args) print_help(); os.exit(0) end
|
||||||
|
FLAG_HANDLERS["--verbose"] = function(args) args.verbose = true end
|
||||||
|
|
||||||
FLAG_HANDLERS["--source"] = function(args, argv, arg_idx)
|
FLAG_HANDLERS["--source"] = function(args, argv, arg_idx)
|
||||||
local value, value_idx = require_flag_value(argv, arg_idx, "--source")
|
local value, value_idx = require_flag_value(argv, arg_idx, "--source")
|
||||||
args.sources[#args.sources + 1] = value
|
args.sources[#args.sources + 1] = value
|
||||||
@@ -490,14 +490,13 @@ end
|
|||||||
--- @param args ParsedArgs
|
--- @param args ParsedArgs
|
||||||
--- @return PassCtx
|
--- @return PassCtx
|
||||||
local function build_ctx(args)
|
local function build_ctx(args)
|
||||||
local normalized_project_root = duffle.normalize_path(args.project_root)
|
local normalized_project_root = duffle.normalize_path(args.project_root)
|
||||||
local project_root = normalized_project_root
|
local project_root = normalized_project_root
|
||||||
local project_root_is_absolute = normalized_project_root:match("^%a:/")
|
local project_root_is_absolute = normalized_project_root:match("^%a:/")
|
||||||
or normalized_project_root:sub(1, 2) == "//"
|
or normalized_project_root:sub(1, 2) == "//"
|
||||||
or normalized_project_root:sub(1, 1) == "/"
|
or normalized_project_root:sub(1, 1) == "/"
|
||||||
if not project_root_is_absolute then
|
if not project_root_is_absolute then
|
||||||
-- canonical_path_key validates ordinary relative paths and rejects
|
-- canonical_path_key validates ordinary relative paths and rejects drive-relative paths before the absolute-path rewrite is performed.
|
||||||
-- drive-relative paths before the absolute-path rewrite is performed.
|
|
||||||
duffle.canonical_path_key(normalized_project_root)
|
duffle.canonical_path_key(normalized_project_root)
|
||||||
project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root))
|
project_root = duffle.normalize_path(duffle.to_absolute_path(normalized_project_root))
|
||||||
else
|
else
|
||||||
@@ -511,8 +510,7 @@ local function build_ctx(args)
|
|||||||
project_root = project_root,
|
project_root = project_root,
|
||||||
})
|
})
|
||||||
if not ok_resolve then
|
if not ok_resolve then
|
||||||
io.stderr:write("ps1_meta: cannot resolve --unity-root "
|
io.stderr:write("ps1_meta: cannot resolve --unity-root " .. tostring(args.unity_root) .. ": " .. tostring(resolved) .. "\n")
|
||||||
.. tostring(args.unity_root) .. ": " .. tostring(resolved) .. "\n")
|
|
||||||
os.exit(EXIT_INTERNAL_ERROR)
|
os.exit(EXIT_INTERNAL_ERROR)
|
||||||
end
|
end
|
||||||
resolution = resolved
|
resolution = resolved
|
||||||
@@ -528,8 +526,7 @@ local function build_ctx(args)
|
|||||||
local path = duffle.normalize_path(input_path)
|
local path = duffle.normalize_path(input_path)
|
||||||
local key_ok, key_or_error = pcall(duffle.canonical_path_key, path)
|
local key_ok, key_or_error = pcall(duffle.canonical_path_key, path)
|
||||||
if not key_ok then
|
if not key_ok then
|
||||||
error("ps1_meta: invalid --source " .. input_path .. ": "
|
error("ps1_meta: invalid --source " .. input_path .. ": " .. tostring(key_or_error), 0)
|
||||||
.. tostring(key_or_error), 0)
|
|
||||||
end
|
end
|
||||||
local file = io.open(path, "r")
|
local file = io.open(path, "r")
|
||||||
if not file then
|
if not file then
|
||||||
@@ -627,9 +624,7 @@ local function topo_sort(passes, requested_set)
|
|||||||
changed = false
|
changed = false
|
||||||
for name, _ in pairs(needed) do
|
for name, _ in pairs(needed) do
|
||||||
local pass = passes[name]
|
local pass = passes[name]
|
||||||
if not pass then
|
if not pass then error("unknown pass '" .. name .. "' requested") end
|
||||||
error("unknown pass '" .. name .. "' requested")
|
|
||||||
end
|
|
||||||
for _, dep in ipairs(pass.deps) do
|
for _, dep in ipairs(pass.deps) do
|
||||||
if not needed[dep] then
|
if not needed[dep] then
|
||||||
needed[dep] = true
|
needed[dep] = true
|
||||||
|
|||||||
+135
-6
@@ -14,16 +14,21 @@ $url_armips = 'https://github.com/Kingcom/armips.git'
|
|||||||
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
$url_pcsx_redux = 'https://github.com/grumpycoders/pcsx-redux.git'
|
||||||
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
$url_psyq_iwyu = 'https://github.com/johnbaumann/psyq_include_what_you_use.git'
|
||||||
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
$url_lpeg = 'https://github.com/roberto-ieru/LPeg.git'
|
||||||
|
# $url_mkpsxiso = 'https://github.com/Lameguy64/mkpsxiso.git'
|
||||||
|
|
||||||
|
$url_mkpsxiso_win64 = 'https://github.com/Lameguy64/mkpsxiso/releases/download/v2.30/mkpsxiso-2.30-win64.zip'
|
||||||
|
|
||||||
$path_armips = join-path $path_toolchain 'armips'
|
$path_armips = join-path $path_toolchain 'armips'
|
||||||
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
$path_pcsx_redux = join-path $path_toolchain 'pcsx-redux'
|
||||||
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
$path_psyq_iwyu = join-path $path_toolchain 'psyq_iwyu'
|
||||||
$path_lpeg = join-path $path_toolchain 'lpeg'
|
$path_lpeg = join-path $path_toolchain 'lpeg'
|
||||||
|
$path_mkpsxiso = join-path $path_toolchain 'mkpsxiso'
|
||||||
|
|
||||||
clone-gitrepo $path_armips $url_armips
|
clone-gitrepo $path_armips $url_armips
|
||||||
clone-gitrepo $path_lpeg $url_lpeg
|
clone-gitrepo $path_lpeg $url_lpeg
|
||||||
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
clone-gitrepo $path_pcsx_redux $url_pcsx_redux
|
||||||
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
clone-gitrepo $path_psyq_iwyu $url_psyq_iwyu
|
||||||
|
# clone-gitrepo $path_mkpsxiso $url_mkpsxiso
|
||||||
|
|
||||||
$path_armips_build = join-path $path_armips 'build'
|
$path_armips_build = join-path $path_armips 'build'
|
||||||
verify-path $path_armips_build
|
verify-path $path_armips_build
|
||||||
@@ -56,7 +61,120 @@ if (-not $msbuild_exe) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
$path_pcsx_sln = join-path $path_pcsx_redux 'vsprojects\pcsx-redux.sln'
|
||||||
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
# NuGet restore — required before MSBuild.
|
||||||
|
# pcsx-redux's .vcxproj files use the legacy packages.config style with
|
||||||
|
# hardcoded `<Import Project="..\packages\{id}.{ver}\...">` directives.
|
||||||
|
# MSBuild's `/t:Restore` won't fetch missing packages here (the local
|
||||||
|
# packages\ dir is checked but no package-source lookup happens), and
|
||||||
|
# `dotnet restore` errors on packages.config projects, so we walk every
|
||||||
|
# packages.config, parse out the <package id version/> entries, and pull
|
||||||
|
# any missing .nupkg directly from api.nuget.org's flat container.
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
$path_pcsx_packages = join-path $path_pcsx_redux 'vsprojects\packages'
|
||||||
|
$nuget_flat_container = 'https://api.nuget.org/v3-flatcontainer'
|
||||||
|
|
||||||
|
# Collect required (id, version) pairs from every packages.config.
|
||||||
|
$required_packages = @{}
|
||||||
|
Get-ChildItem -Path (join-path $path_pcsx_redux 'vsprojects') -Filter 'packages.config' -Recurse -ErrorAction SilentlyContinue |
|
||||||
|
ForEach-Object {
|
||||||
|
[xml]$xml = Get-Content -LiteralPath $_.FullName -Raw
|
||||||
|
foreach ($pkg in $xml.packages.package) {
|
||||||
|
$key = '{0}|{1}' -f $pkg.id, $pkg.version
|
||||||
|
$required_packages[$key] = @{ id = $pkg.id; version = $pkg.version }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# Ensure the packages root exists.
|
||||||
|
if (-not (Test-Path -LiteralPath $path_pcsx_packages)) {
|
||||||
|
New-Item -ItemType Directory -Path $path_pcsx_packages -Force | Out-Null
|
||||||
|
}
|
||||||
|
|
||||||
|
# Download anything missing. Skip the package entirely if its dir already has
|
||||||
|
# any contents (the legacy packages.config style means the targets file
|
||||||
|
# location varies per package — `luajit.native` puts it at build/native/,
|
||||||
|
# `glfw` puts it elsewhere — so we can't probe a specific path; just check
|
||||||
|
# whether the dir is non-empty).
|
||||||
|
Add-Type -AssemblyName System.IO.Compression.FileSystem
|
||||||
|
foreach ($pkg in $required_packages.Values) {
|
||||||
|
$pkgDir = Join-Path $path_pcsx_packages ('{0}.{1}' -f $pkg.id, $pkg.version)
|
||||||
|
if ((Test-Path -LiteralPath $pkgDir) -and `
|
||||||
|
(@(Get-ChildItem -LiteralPath $pkgDir -Recurse -ErrorAction SilentlyContinue).Count -gt 0)) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
$url = '{0}/{1}/{2}/{1}.{2}.nupkg' -f $nuget_flat_container, $pkg.id, $pkg.version
|
||||||
|
$nupkg = Join-Path $pkgDir ('{0}.{1}.nupkg' -f $pkg.id, $pkg.version)
|
||||||
|
New-Item -ItemType Directory -Path $pkgDir -Force | Out-Null
|
||||||
|
Write-Host "Fetching NuGet package: $($pkg.id) $($pkg.version)"
|
||||||
|
try {
|
||||||
|
Invoke-WebRequest -Uri $url -OutFile $nupkg -UseBasicParsing -ErrorAction Stop
|
||||||
|
[System.IO.Compression.ZipFile]::ExtractToDirectory($nupkg, $pkgDir)
|
||||||
|
Remove-Item -LiteralPath $nupkg -Force
|
||||||
|
} catch {
|
||||||
|
$msg = $_.Exception.Message
|
||||||
|
if ($msg -match '404') {
|
||||||
|
Write-Host " Not on nuget.org (vendored?) — skipping $url"
|
||||||
|
} else {
|
||||||
|
Write-Warning "Failed to fetch $url — $msg"
|
||||||
|
}
|
||||||
|
if (Test-Path -LiteralPath $nupkg) { Remove-Item -LiteralPath $nupkg -Force }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
# isoffi.lua size guard — `core.vcxproj` #includes src/core/isoffi.lua into
|
||||||
|
# luaiso.cc via the `-- lualoader, R"EOF(...)EOF"` trick. The raw string
|
||||||
|
# literal between R"EOF(-- and -- )EOF" must stay under ~16,379 bytes or
|
||||||
|
# MSVC (19.44) fails with C2026 (its actual raw-string limit is 16,384,
|
||||||
|
# minus 5 bytes for the `-- lualoader, ` prefix). If the upstream file
|
||||||
|
# grows past that, trim it: remove license header, trailing whitespace,
|
||||||
|
# blank separators, inline comments, and shrink 4-space indent to 2-space.
|
||||||
|
# Idempotent — only writes when the raw string exceeds the limit.
|
||||||
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
$path_isoffi = join-path $path_pcsx_redux 'src\core\isoffi.lua'
|
||||||
|
if (Test-Path -LiteralPath $path_isoffi) {
|
||||||
|
$content = Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8
|
||||||
|
$startMarker = $content.IndexOf('R"EOF(--')
|
||||||
|
$endMarker = $content.IndexOf('-- )EOF"')
|
||||||
|
$literalLen = if ($startMarker -ge 0 -and $endMarker -gt $startMarker) {
|
||||||
|
$endMarker - ($startMarker + 8)
|
||||||
|
} else { -1 }
|
||||||
|
# Effective MSVC raw-string limit for the lualoader prefix is 16379 bytes.
|
||||||
|
if ($literalLen -gt 16379) {
|
||||||
|
Write-Host "isoffi.lua raw string is $literalLen bytes (>16379); trimming for MSVC C2026 limit."
|
||||||
|
$lines = $content -split "`n"
|
||||||
|
$markerIdx = -1
|
||||||
|
for ($i = 0; $i -lt $lines.Length; $i++) {
|
||||||
|
if ($lines[$i] -match '^-- \)EOF"') { $markerIdx = $i; break }
|
||||||
|
}
|
||||||
|
$newLines = @()
|
||||||
|
for ($i = 0; $i -lt $lines.Length; $i++) {
|
||||||
|
$lineNum = $i + 1
|
||||||
|
$line = $lines[$i]
|
||||||
|
# Keep the first line and the EOF-marker line untouched.
|
||||||
|
if ($i -eq 0 -or $i -eq $markerIdx) { $newLines += $line; continue }
|
||||||
|
# Drop the GPL license header (lines 2-17).
|
||||||
|
if ($lineNum -ge 2 -and $lineNum -le 17) { continue }
|
||||||
|
# Drop blank separator lines.
|
||||||
|
if ($line -match '^\s*$') { continue }
|
||||||
|
# Drop trailing whitespace.
|
||||||
|
$line = $line -replace '\s+$', ''
|
||||||
|
# Drop inline comments (anything from `--` to end of line).
|
||||||
|
$line = $line -replace '\s*--.*$', ''
|
||||||
|
# Shrink 4-space indent to 2-space.
|
||||||
|
$line = $line -replace '^( )', ' '
|
||||||
|
if ($line -match '^\s*$') { continue }
|
||||||
|
$newLines += $line
|
||||||
|
}
|
||||||
|
($newLines -join "`n") | Out-File -LiteralPath $path_isoffi -Encoding utf8 -NoNewline
|
||||||
|
$newLen = ((Get-Content -LiteralPath $path_isoffi -Raw -Encoding utf8) `
|
||||||
|
-replace '.*R"EOF\(--', '' -replace '-- \)EOF".*', '').Length
|
||||||
|
Write-Host "isoffi.lua trimmed: $literalLen -> $newLen bytes of raw string content."
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
& $msbuild_exe $path_pcsx_sln /p:Configuration=Release /p:Platform=x64 /p:PlatformToolset=v143 /m /v:minimal
|
||||||
|
|
||||||
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
# Locate luajit via scoop. `luajit.exe` is on PATH via scoop's shim;
|
||||||
# we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers).
|
# we use `scoop prefix` to find the install root for the include dir (needed to compile lpeg against luajit's headers).
|
||||||
@@ -70,8 +188,8 @@ if (-not $luajit_prefix -or -not (Test-Path (Join-Path $luajit_prefix 'bin/luaji
|
|||||||
# Discover the luajit include dir by globbing `include/luajit-*`.
|
# Discover the luajit include dir by globbing `include/luajit-*`.
|
||||||
# This avoids hardcoding a specific version (e.g. `luajit-2.1`).
|
# This avoids hardcoding a specific version (e.g. `luajit-2.1`).
|
||||||
$luajit_include_root = Join-Path $luajit_prefix 'include'
|
$luajit_include_root = Join-Path $luajit_prefix 'include'
|
||||||
$lua_inc_dir = Get-ChildItem -Path $luajit_include_root -Directory -Filter 'luajit-*' -ErrorAction SilentlyContinue |
|
$lua_inc_dir = Get-ChildItem -Path $luajit_include_root -Directory -Filter 'luajit-*' -ErrorAction SilentlyContinue |
|
||||||
Select-Object -First 1 -ExpandProperty FullName
|
Select-Object -First 1 -ExpandProperty FullName
|
||||||
if (-not $lua_inc_dir) {
|
if (-not $lua_inc_dir) {
|
||||||
write-error "No 'luajit-*' include dir found under '$luajit_include_root'. The scoop luajit install may be broken."
|
write-error "No 'luajit-*' include dir found under '$luajit_include_root'. The scoop luajit install may be broken."
|
||||||
exit 1
|
exit 1
|
||||||
@@ -90,7 +208,7 @@ $lpeg_compile_args = @(
|
|||||||
'-o', 'lpeg.dll'
|
'-o', 'lpeg.dll'
|
||||||
) + $lpeg_sources + @('-lluajit-5.1')
|
) + $lpeg_sources + @('-lluajit-5.1')
|
||||||
push-location $path_lpeg
|
push-location $path_lpeg
|
||||||
& gcc @lpeg_compile_args
|
& gcc @lpeg_compile_args
|
||||||
pop-location
|
pop-location
|
||||||
|
|
||||||
# ════════════════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
@@ -101,8 +219,8 @@ pop-location
|
|||||||
|
|
||||||
$path_lfs = join-path $path_toolchain 'lfs'
|
$path_lfs = join-path $path_toolchain 'lfs'
|
||||||
verify-path $path_lfs
|
verify-path $path_lfs
|
||||||
$lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c'
|
$lfs_src = join-path $path_pcsx_redux 'third_party\luafilesystem\src\lfs.c'
|
||||||
$lfs_dll = join-path $path_lfs 'lfs.dll'
|
$lfs_dll = join-path $path_lfs 'lfs.dll'
|
||||||
$lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
$lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
||||||
& gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import
|
& gcc -O2 -shared "-I$lua_inc_dir" -o $lfs_dll $lfs_src $lfs_dll_import
|
||||||
|
|
||||||
@@ -112,6 +230,17 @@ $lfs_dll_import = join-path $luajit_lib_dir 'libluajit-5.1.dll.a'
|
|||||||
# ════════════════════════════════════════════════════════════════════════════
|
# ════════════════════════════════════════════════════════════════════════════
|
||||||
|
|
||||||
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
$path_openbios = join-path $path_pcsx_redux 'src\mips\openbios'
|
||||||
|
|
||||||
|
# Wipe stale *.dep files across src\mips. These cache absolute paths to the
|
||||||
|
# GCC headers directory; if the toolchain was upgraded (e.g. v14.2.0 → v16.1.0)
|
||||||
|
# Make reads the stale paths and aborts with "no rule to make target .../stddef.h".
|
||||||
|
# `make clean` in openbios only clears its own dir — subdirs like
|
||||||
|
# common/crt0/, modplayer/, and shell/ keep their stale .dep files. Easier to
|
||||||
|
# just delete the lot before each build than to teach every Makefile about
|
||||||
|
# deepclean recursion.
|
||||||
|
Get-ChildItem -Path (join-path $path_pcsx_redux 'src\mips') -Recurse -Filter '*.dep' -ErrorAction SilentlyContinue |
|
||||||
|
ForEach-Object { Remove-Item -LiteralPath $_.FullName -Force }
|
||||||
|
|
||||||
push-location $path_openbios
|
push-location $path_openbios
|
||||||
& make clean
|
& make clean
|
||||||
& make
|
& make
|
||||||
|
|||||||
Reference in New Issue
Block a user